[
 {
  "corpusId": "286775791",
  "title": "CarePilot: A Multi-Agent Framework for Long-Horizon Computer Task Automation in Healthcare",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 6,
  "publication_date": "2026-03-25",
  "months_since_pub": 6,
  "citations_per_month": 1.0,
  "artifact_name": "CareFlow (benchmark) / CarePilot (agent framework)",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete each of 8-24 consecutive GUI decisions/actions required by one long-horizon clinical software workflow (e.g. DICOM viewer, EHR, lab information system); correctly ground next-action predictions in the current visual interface and system state throughout the workflow",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each of the 8-24 consecutive decisions in a workflow builds on the system/interface state left by the prior decision, so an incorrect earlier action in medical annotation tools, DICOM viewers, EHR systems, or laboratory information systems can invalidate or block later steps in the same task.",
  "n_goals": "each task comprises 8-24 consecutive decisions, spanning four major clinical-software categories (DICOM viewing/infrastructure, medical image annotation, hospital EMR systems, laboratory information systems)",
  "tracking_demand": "The agent must track the current visual interface state and prior decisions across 8-24 consecutive steps of a clinical software workflow, using dual-memory (long-term and short-term experience) to predict the next semantic action correctly.",
  "scoring": "other:actor-critic-iterative-refinement -- the Critic evaluates each action, updates memory, and either executes or provides corrective feedback, i.e. the evaluation architecture operates at the per-action level even though the paper reports an aggregate benchmark performance percentage rather than an explicit points-based subgoal rubric.",
  "horizon_value": "each task comprises 8-24 consecutive decisions/steps",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "CarePilot achieves state-of-the-art performance, outperforming strong closed-source and open-source multimodal baselines by approximately 15.26% and 3.38% respectively on CareFlow and an out-of-distribution dataset; no human-expert baseline comparison is reported.",
  "availability": null,
  "goal_span": "we introduce CareFlow, a high-quality human-annotated benchmark comprising complex, long-horizon software workflows across medical annotation tools, DICOM viewers, EHR systems, and laboratory information systems.",
  "horizon_span": "each task pairs a natural-language goal with GUI screenshots representing authentic clinical workflows... 8-24 consecutive decisions",
  "extraction_confidence": 2,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286775791",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289623133",
  "title": "ChainWorld: Composing Long-Horizon Desktop Workloads from Atomic OSWorld Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-06-19",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "ChainWorld",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a chain of 2 to 4 sequentially composed atomic OSWorld desktop tasks while sustaining state across all of them; under single-turn evaluation, complete the whole chain from one combined prompt; under multi-turn evaluation, complete each task as it is revealed one at a time while retaining session state",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later tasks in a chain depend on state left behind by earlier tasks (chains are built via directional compatibility search over compatible orderings), and multi-turn evaluation adds session-continuity demands across turns.",
  "n_goals": "347 chains, each of length two to four atomic tasks",
  "tracking_demand": "Agent must sustain desktop application/file state across multiple chained objectives and, in multi-turn mode, manage session continuity as tasks are revealed one at a time.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": "two to four",
  "horizon_unit": "other:atomic-tasks-per-chain",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per composed chain/workload, not per single atomic OSWorld task)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "ChainWorld, which composes atomic OSWorld tasks into long horizon desktop workloads through directional compatibility search while preserving the source evaluators. The resulting workload contains 347 chains of length two to four",
  "horizon_span": "The resulting workload contains 347 chains of length two to four and compares two renderings of the same task sequence.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289623133",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288650865",
  "title": "CutVerse: A Compositional GUI Agents Benchmark for Media Post-Production Editing",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-05-19",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "CutVerse",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a long-horizon, compositional media-editing task grounded in an authentic editing workflow using one of 7 professional applications (e.g., Premiere Pro, Photoshop); correctly execute tightly coupled, dense multimodal interaction sequences within the application's GUI",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Editing tasks are 'tightly coupled interaction sequences', so a compositional GUI action later in the sequence depends on the state established by dense multimodal interface interactions earlier (e.g., a layer or clip selected in a prior step).",
  "n_goals": "186 complex, long-horizon tasks across 7 professional applications",
  "tracking_demand": "Agent must track compositional GUI state (selected layers/clips/tools) across a tightly coupled, dense-interface interaction sequence throughout a long-horizon editing task.",
  "scoring": "binary-final-success - existing agents achieve only 36.0% task success on realistic media editing tasks; abstract does not describe an explicit subgoal-checkpoint partial-credit scheme, framing results as overall task success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Existing agents achieve only 36.0% task success on realistic media editing tasks; models show promising spatial grounding and multimodal alignment but remain limited in long-horizon reliability and domain-specific planning; no human/expert baseline is given.",
  "availability": null,
  "goal_span": "We curate expert demonstrations across 7 professional applications (e.g., Premiere Pro, Photoshop), covering 186 complex, long-horizon tasks grounded in authentic editing workflows, involving dense multimodal interfaces and tightly coupled interaction sequences.",
  "horizon_span": "covering 186 complex, long-horizon tasks grounded in authentic editing workflows, involving dense multimodal interfaces and tightly coupled interaction sequences.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288650865",
  "provenance": "forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288863121",
  "title": "DeskCraft: Benchmarking Desktop Agents on Professional Workflows and Human-in-the-Loop Collaboration",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 2,
  "publication_date": "2026-06-02",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "DeskCraft",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete long-horizon professional creative/engineering workflows (design, video, audio, 3D creation) requiring over 50 execution steps; proactively seek necessary information from the user under uncertainty (agent-initiated clarification); correctly handle user-initiated interruptions during execution and incorporate post-turn feedback after signaling completion",
  "goal_origin": "mixed:task-given-upfront-with-human-in-the-loop-interruptions-and-clarifications-injected-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Because tasks require over 50 execution steps in professional creative/engineering software, later steps depend on cumulative in-progress work, and mid-turn human interruptions or clarifications must be integrated into the ongoing plan without discarding already-completed steps; post-turn feedback after a signaled completion may require revisiting prior completed work.",
  "n_goals": "538 tasks; a multilevel difficulty taxonomy including long-horizon tasks requiring over 50 execution steps; interaction protocol spans mid-turn (agent-initiated clarification, user-initiated interruption) and post-turn (user feedback) exchanges",
  "tracking_demand": "The agent must track its own multi-step progress (over 50 execution steps for long-horizon tasks) within professional creative/engineering software, while also tracking pending clarification needs, handling user interruptions mid-execution, and incorporating post-completion feedback into further revisions.",
  "scoring": "binary-final-success (per task) -- GPT-5.4 reaches 31.6% on standard tasks and 27.6% on interactive tasks, framed as task-level completion percentages; the abstract does not describe an explicit subgoal-checkpoint partial-credit rubric distinct from these completion rates, though a multilevel difficulty taxonomy implies graded task difficulty.",
  "horizon_value": "long-horizon tasks require over 50 execution steps",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "GPT-5.4 reaches 31.6% on standard tasks and 27.6% on interactive tasks; persistent failures found in long-horizon workflow delivery and proactive clarification. No human/expert baseline is reported.",
  "availability": "https://github.com/mrwwk/DeskCraft",
  "goal_span": "DeskCraft organizes tasks into a multilevel difficulty taxonomy, with long horizon tasks requiring over 50 execution steps, and covers professional creative software across design, video, audio, and 3D creation. Furthermore, DeskCraft formalizes human-agent collaboration into an interaction protocol covering mid-turn and post-turn exchanges. Mid-turn interaction captures both agent-initiated clarification under uncertainty and user-initiated interruption during execution, while post-turn interaction accommodates user-driven feedback after the agent signals completion, together spanning the full space of realistic collaboration patterns.",
  "horizon_span": "DeskCraft organizes tasks into a multilevel difficulty taxonomy, with long horizon tasks requiring over 50 execution steps, and covers professional creative software across design, video, audio, and 3D creation.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288863121",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290179063",
  "title": "DevicesWorld: Benchmarking Cross-Device Agents in Heterogeneous Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-07-15",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "DevicesWorld",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "acquire information on one device (e.g. phone) needed to complete a task; process/transform that information on a different device (e.g. desktop); deliver or display the final result on yet another device; satisfy cross-device dependencies and rule-based verifiers across the whole task",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each task requires 'end-to-end tasks with cross-device dependencies,' so acquiring information on one device is a prerequisite for the processing step on another device, and the final device's output depends on correctly completed steps on the earlier devices; agents 'confuse source and output devices' or 'terminate before all conditions are jointly satisfied.'",
  "n_goals": "6,140 tasks integrating 3 device-environment classes (mobile, desktop, IoT)",
  "tracking_demand": "The agent must track which device holds which piece of needed information, what has already been acquired/processed/delivered across devices, and whether all task conditions (verified from device states and generated files) are jointly satisfied.",
  "scoring": "other:joint-condition-verification. 'About 28.7% [of failed runs] satisfy at least one scoring condition yet still fail the full task,' i.e. the paper explicitly reports partial condition-satisfaction distinct from full-task success -- a form of subgoal-level tracking even though the headline metric is full end-to-end success.",
  "horizon_value": "each task permits at most 50 interaction steps",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (maximum step budget)",
  "horizon_stated": "yes",
  "headline_result": "All evaluated methods achieve low success rates, with the best reaching only 12.5%; no human/expert baseline given.",
  "availability": null,
  "goal_span": "We introduce DevicesWorld, a large-scale executable benchmark for cross-device collaborative operation. DevicesWorld contains 6,140 tasks and integrates three classes of device environments -- mobile, desktop, and IoT -- into a unified cross-device interaction and evaluation framework... Among failed runs, about 28.7% satisfy at least one scoring condition yet still fail the full task.",
  "horizon_span": "each task permits at most 50 interaction steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290179063",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288741510",
  "title": "GUITestScape: Towards Open-set Evaluation on Exploratory GUI Testing",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "GUITestScape",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "autonomously navigate an app to discover defects with no predefined test script; distinguish and separately diagnose interaction defects versus display defects; decompose the testing trajectory into independently diagnosable capabilities rather than a single end-state judgment",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Because there is no predefined test script, the agent's later exploration decisions depend on what parts of the app and what defect types it has already investigated in the same session; GUIJudge explicitly decomposes 'an agent's testing trajectory into independently diagnosable capabilities,' meaning the overall testing session is evaluated as a process of several distinguishable sub-goals rather than one final verdict.",
  "n_goals": "61 real-world Android applications; 508 preset defects spanning interaction and display types",
  "tracking_demand": "The agent must track which parts of an app it has already explored and which candidate defects (interaction or display) it has already investigated, to continue open-ended exploration without redundant re-testing or missing an area.",
  "scoring": "other:process-aware-capability-decomposition. GUIJudge is 'an open-set evaluator that decomposes an agent's testing trajectory into independently diagnosable capabilities,' explicit process-level, multi-capability credit rather than a single end-state pass/fail judgment tied only to predefined defect annotations.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "current evaluation falls short on two fronts... evaluation protocols are bound to predefined defect annotations, collapsing the testing process into a single end-state judgment that conflates qualitatively distinct failure modes. To address these challenges, we present GUITestScape, an interactive benchmark covering 61 real-world Android applications and 508 preset defects spanning interaction and display types, and introduce GUIJudge, an open-set evaluator that decomposes an agent's testing trajectory into independently diagnosable capabilities.",
  "horizon_span": "GUITestScape, an interactive benchmark covering 61 real-world Android applications and 508 preset defects spanning interaction and display types",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288741510",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289622568",
  "title": "Beyond Global Replanning: Hierarchical Recovery for Cross-Device Agent Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-06-18",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "HeraBench",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "decompose a cross-device task and assign subtasks across heterogeneous devices via unified API-CLI-GUI execution; recover from injected device-local strategy failures without escalation; escalate to orchestrator-level global replanning when a failure exceeds device-local recovery scope; complete an end-to-end cross-device workflow over Linux and Android devices",
  "goal_origin": "mixed:workflow-given-up-front-failures-injected-by-environment",
  "decomposition": "hierarchical",
  "interdependence": "A failure at the device-local strategy level may be repairable without affecting the global plan, but if not, it must propagate up to orchestrator-level replanning, so the choice of recovery scope is coupled across the two tiers and across device boundaries.",
  "n_goals": null,
  "tracking_demand": "The system must track per-device execution-strategy state, a compact cross-layer failure abstraction distinguishing device-local from global failure scope, and overall workflow progress across Linux and Android devices.",
  "scoring": "other:completion-instruction-adherence-perfect-pass-and-token-cost. The paper reports completion rate, instruction adherence, and perfect-pass rate together with token cost as separate graded dimensions, though no explicit per-subgoal checkpoint credit is described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "H-RePlan substantially outperforms single-strategy and coarse-grained multi-device baselines on HeraBench, achieving higher completion, instruction adherence, and perfect-pass rates while reducing token cost; no human/expert baseline given.",
  "availability": null,
  "goal_span": "we introduce HeraBench, a fault-injected benchmark that constructs cross-device workflows over Linux and Android devices and injects strategy- and device-level failures.",
  "horizon_span": "HeraBench, a fault-injected benchmark that constructs cross-device workflows over Linux and Android devices and injects strategy- and device-level failures.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289622568",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291876338",
  "title": "JarvisGUI: Towards Cross-Device GUI Agents with Dynamic Task Composition",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-09-09",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "JarvisGUI",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "transfer intermediate results correctly between two or more heterogeneous devices/platforms; maintain shared state consistently across Android, Windows, and Ubuntu platforms; compose a multi-step, cross-device workflow from input-output-typed GUI sub-tasks; complete a workflow requiring four or more chained subtasks despite compounding dependency-tracking difficulty",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "GUI tasks are formulated as typed input-output transformations that compose into multi-step, cross-device workflows, so a later subtask's inputs depend on outputs correctly transferred and maintained from an earlier subtask on a different device; the paper reports success collapsing to near zero specifically as subtask count (>=4) grows, evidencing this compounding dependency.",
  "n_goals": "dynamically composed workflows; success drops to near zero for tasks requiring four or more subtasks",
  "tracking_demand": "The agent must track shared state and intermediate results as they transfer across heterogeneous devices/platforms, verifying that dependency steps (e.g. file transfer, renaming) have actually executed rather than being falsely treated as completed.",
  "scoring": "other:task-success-rate-by-subtask-count",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "State-of-the-art open-source GUI agents struggle with state-transfer awareness and long-horizon dependency management; task success rate drops to near zero for tasks requiring 4+ subtasks across all evaluated models; no human baseline given.",
  "availability": null,
  "goal_span": "JarvisGUI formulates GUI tasks as input-output transformations under a lightweight type system, which allows us to automatically compose multi-step, cross-device workflows and dynamically evaluate agent performance within a unified framework. By evaluating agents in virtual environments spanning multiple operating systems, JarvisGUI reveals that state-of-the-art open-source GUI agents struggle with the state-transfer awareness, cross-platform contextual reasoning, and long-horizon dependency management required for real-world workflows",
  "horizon_span": "The task success rate exhibits a sharp decline as the number of subtasks increases, dropping to near zero for tasks requiring four or more subtasks across all evaluated models.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291876338",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289622864",
  "title": "MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 2,
  "publication_date": "2026-06-21",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "MacAgentBench",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a multi-application task on real macOS desktop software using both GUI and CLI interaction; reach each of several fine-grained, capability-annotated checkpoints within a multi-application task",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Multi-application tasks require coordinating actions across multiple macOS apps (676 tasks across 25 applications), and 'fine-grained multi-checkpoint scoring with capability annotations' implies checkpoints within a task are ordered/related stages of the same multi-application workflow.",
  "n_goals": "676 tasks across 25 applications (~60% involving both GUI and CLI interaction)",
  "tracking_demand": "Agent must track partial progress across multiple checkpoints within a multi-application task, since 'models with similar Pass@1 can differ substantially in sub-goal completion' according to the fine-grained metrics.",
  "scoring": "subgoal-checkpoint-partial-credit \u2014 explicitly, 'fine-grained multi-checkpoint scoring with capability annotations for multi-application tasks', and 'models with similar Pass@1 can differ substantially in sub-goal completion', confirming distinct subgoal-level credit beyond binary Pass@1.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best configuration, Claude Opus 4.6 on OpenClaw, attains 73.7% Pass@1, though this advantage is 'primarily driven by the skill library rather than by framework design' (no human baseline given).",
  "availability": "https://github.com/JetAstra/MacAgentBench",
  "goal_span": "The benchmark adopts deterministic rule-based evaluation and introduces fine-grained multi-checkpoint scoring with capability annotations for multi-application tasks.",
  "horizon_span": "We present MacAgentBench, a comprehensive macOS agent benchmark comprising 676 tasks across 25 applications, with nearly 60% involving both GUI and CLI interaction",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289622864",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288863349",
  "title": "MedCUA-Bench: A Screenshot-Only Benchmark for Clinical Computer-Use Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 1,
  "publication_date": "2026-06-02",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "MedCUA-Bench",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete clinical computer-use scenarios across 10 medical domains; satisfy paired intent-level and step-level goals within each task; avoid violations across five clinical safety dimensions while completing the task",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Step-level UI execution must remain faithful to the higher-level clinical intent, and a task can be judged unsuccessful even with surface completion if any of the five safety dimensions is violated.",
  "n_goals": "18 clinical scenarios across 10 medical domains, each with paired intent- and step-level goals plus 5 safety dimensions",
  "tracking_demand": "Agent must track progress toward the high-level clinical intent, the individual UI steps needed to realize it, and continuously check five clinical safety dimensions across the task.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best closed-source model reaches 54.2% strict success overall but under 9% on the real OpenEMR system; open-source agents average only 2.5% (best 16.2%); no human baseline reported.",
  "availability": null,
  "goal_span": "Each task ships with paired intent- and step-level goals to disentangle clinical reasoning from UI execution, and is evaluated by a deterministic checker over task completion and five clinical safety dimensions.",
  "horizon_span": "It covers 18 clinical scenarios across 10 medical domains, reconstructed from real product manuals and open-source medical systems to capture authentic clinical interfaces",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288863349",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289299570",
  "title": "MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 1,
  "publication_date": "2026-06-15",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "MyPCBench",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a personal-assistant task requiring the agent's own accumulated context/history; operate correctly across 17 simulated real-world web applications seeded for one canonical persona; coordinate actions that span many applications on the same Linux desktop; sustain a long trajectory without abandoning early or looping unproductively",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are inspired by real personal-assistant requests and are seeded to one canonical persona, so completing a task correctly requires consistent use of that persona's context/history across 17 interlinked applications; failures cluster specifically on tasks spanning many applications and long trajectories.",
  "n_goals": "184 tasks across 17 simulated applications",
  "tracking_demand": "The agent must track the canonical persona's context, historical data, and logged-in-account state across a full Linux desktop stack, coordinating GUI and bash actions across many applications without unproductive step looping.",
  "scoring": "other:full-task-solve-rate-per-model",
  "horizon_value": "mean 22-31 steps (GPT family, Sonnet); mean 52-85 steps (Opus, Qwen models)",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task (mean executed GUI/bash actions, model-dependent)",
  "horizon_stated": "yes",
  "headline_result": "Best model, Claude Opus 4.6, fully solves 55.4% of tasks, the only model above 50%; no human baseline given.",
  "availability": "https://mypcbench.com",
  "goal_span": "We introduce MyPCBench, which tests computer-use agents as personal assistants on a Linux desktop populated with 17 simulated real-world web applications and a full desktop stack, all seeded for one canonical persona, Michael Scott from The Office. We define 184 tasks in this environment ... Model failures cluster on tasks that span many applications and on long trajectories, where personalization stresses an assistant the most.",
  "horizon_span": "The GPT family and Sonnet abandon early (mean 22-31 steps), while Opus and the Qwen models keep working past the point where the rubric is recoverable (mean 52-85 steps).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289299570",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285269029",
  "title": "OS-Marathon: Benchmarking Computer-Use Agents on Long-Horizon Repetitive Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 5,
  "publication_date": null,
  "months_since_pub": 3,
  "citations_per_month": 1.67,
  "artifact_name": "OS-Marathon",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete each of 242 long-horizon, repetitive computer-use tasks across 2 domains (e.g., processing expense reports from receipts, entering grades from exam papers); correctly repeat the same structured sub-workflow logic across many data items within one task; learn workflow logic from a few-shot condensed demonstration and generalize it to larger, unseen data collections",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each long-horizon task consists of many repeated per-item sub-workflows (e.g., one receipt or one exam paper at a time) sharing the same underlying logic, so correctly learning that logic from a few-shot demonstration is a precondition for correctly and consistently repeating it across the full, potentially very large, data collection.",
  "n_goals": "242 tasks across 2 domains; task length scales proportional to the size of the data collection processed",
  "tracking_demand": "The agent must track its position within a long, repetitive workflow (e.g., which receipt or exam paper it is currently processing), correctly apply the learned sub-workflow logic consistently, and avoid drift or errors accumulating across many repeated sub-tasks.",
  "scoring": "other:task-completion-with-repetitive-sub-workflow-consistency. The abstract emphasizes that these tasks extend to extreme lengths proportional to the size of the data to process, implying performance depends on consistent correctness across many repeated sub-items rather than one single short decision, though no explicit per-item partial-credit scheme is stated in the visible abstract.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we establish OS-Marathon, comprising 242 long-horizon, repetitive tasks across 2 domains to evaluate state-of-the-art (SOTA) agents. We then introduce a cost-effective method to construct a condensed demonstration using only few-shot examples to teach agents the underlying workflow logic, enabling them to execute similar workflows effectively on larger, unseen data collections.",
  "horizon_span": "they can extend to extreme lengths proportional to the size of the data to process",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285269029",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291382635",
  "title": "OS-Marathon: Benchmarking Computer-Use Agents on Vast-Horizon, Repetitive Tasks",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2026-01-28",
  "months_since_pub": 8,
  "citations_per_month": 0.0,
  "artifact_name": "OS-Marathon",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete many recurring per-instance sub-workflows within a vast-horizon repetitive task (e.g., process each receipt in a stack); sustain correct execution as length scales with data volume; learn/personalize the recurring sub-workflow logic from a single human demonstration (GraphDemo)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Errors made by a task orchestrator when decomposing the workflow into per-instance subtasks accumulate and propagate across solver agents handling later instances, so naive decomposition compounds failures across the repetitive task.",
  "n_goals": "100 vast-horizon, repetitive tasks across 5 scenarios and 10 domains",
  "tracking_demand": "Agent must correctly and repeatedly apply the same recurring sub-workflow logic across many per-instance items, with execution length scaling with the volume of data to process.",
  "scoring": "other:not-specified-in-abstract",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Vast-horizon, repetitive workflows are common in daily routines, e.g., processing expense reports from a stack of receipts, organising a collection of PDF annotations into structured notes, and are tedious for humans, with execution length scaling with the volume of data to process.",
  "horizon_span": "with execution length scaling with the volume of data to process",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291382635",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "289683715",
  "title": "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 15,
  "publication_date": "2026-06-28",
  "months_since_pub": 3,
  "citations_per_month": 5.0,
  "artifact_name": "OSWorld 2.0",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete each of 108 realistic end-to-end long-horizon computer-use workflows; perform cross-source reasoning across authentic input artifacts; infer implicit state and recover hidden state the task depends on; handle streaming interaction and dynamic environment changes mid-task; operate under safety-sensitive execution constraints (audited separately)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks hinge on hidden state that must be recovered and information that arrives mid-task, so later steps depend on correctly retaining constraints and cross-referencing realistic stateful user-profile data established earlier in the same run; missing or mis-tracked state causes downstream failure.",
  "n_goals": "108 long-horizon workflows; average 318 tool calls per task (vs about 30 in OSWorld 1.0); 500-step completion budget",
  "tracking_demand": "The agent must track constraints, cross-source information, implicit/hidden state, and mid-task updates continuously across an average of 318 tool calls (up to a 500-step budget) per task, rather than resolving everything from the initial instruction alone.",
  "scoring": "other:binary-completion-plus-partial-score",
  "horizon_value": "median about 1.6 hours per task for human users; average of 318 tool calls per task with Claude Opus 4.7 using maximum thinking (vs about 30 in OSWorld 1.0); 500-step completion budget",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (a single end-to-end workflow)",
  "horizon_stated": "yes",
  "headline_result": "Claude Opus 4.8 with maximum thinking completes only 20.6% of tasks (54.8% partial score) under a 500-step budget; GPT-5.5 plateaus near 13%; human users take a median of about 1.6 hours per task, establishing a large agent-vs-human gap.",
  "availability": null,
  "goal_span": "Under our primary binary-completion metric at 500 steps, Claude Opus 4.8 with maximum thinking and batched tool calls scores best but still completes only 20.6% of tasks at a 54.8% partial score; GPT-5.5 is far more token-efficient yet plateaus near 13%.",
  "horizon_span": "Each task represents a realistic end-to-end workflow that takes human users a median of about 1.6 hours to complete and requires an average of 318 tool calls with Claude Opus 4.7 using maximum thinking, compared with about 30 in OSWorld 1.0.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289683715",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286673478",
  "title": "OS-Themis: A Scalable Critic Framework for Generalist GUI Rewards",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 8,
  "publication_date": "2026-03-19",
  "months_since_pub": 6,
  "citations_per_month": 1.33,
  "artifact_name": "OmniGUIRewardBench (OGRBench) / OS-Themis",
  "artifact_kind": "arena/leaderboard",
  "domain": "OS-computer-use",
  "goal_types": "decompose a GUI trajectory into verifiable milestones; audit the evidence chain for each milestone before rendering a final reward verdict; produce a scalable, accurate outcome reward usable for RL training or trajectory filtering",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The final verdict on a trajectory depends on a chain of milestone-level verdicts, each of which must be evidenced before the review mechanism accepts it, so an unverified or contradicted milestone can block the overall reward decision.",
  "n_goals": null,
  "tracking_demand": "The critic must track which milestones in a trajectory have been reached, the evidence supporting each, and audit the full evidence chain before issuing a final reward \u2014 i.e. it tracks reward-relevant progress rather than the acting agent's own task state.",
  "scoring": "milestone-rubric \u2014 trajectories are explicitly decomposed into verifiable milestones with a strict evidence audit before a final verdict, i.e. subgoal-level (milestone) credit is central to the design.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "OS-Themis yields a 10.3% improvement when used to support online RL training on AndroidWorld, and a 6.9% gain for trajectory validation/filtering in self-training (no human baseline given).",
  "availability": null,
  "goal_span": "OS-Themis decomposes trajectories into verifiable milestones to isolate critical evidence for decision making and employs a review mechanism to strictly audit the evidence chain before making the final verdict",
  "horizon_span": "we further introduce OmniGUIRewardBench (OGRBench), a holistic cross-platform benchmark for GUI outcome rewards",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286673478",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287916019",
  "title": "WindowsWorld: A Process-Centric Benchmark of Autonomous GUI Agents in Professional Cross-Application Environments",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 8,
  "publication_date": "2026-04-30",
  "months_since_pub": 5,
  "citations_per_month": 1.6,
  "artifact_name": "WindowsWorld",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete each of a task's average 5.0 sub-goals across up to 17 desktop applications; perform conditional judgment and reasoning that spans 3 or more applications; coordinate a workflow across multiple applications for 78% of tasks that are inherently multi-application",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Cross-application tasks require passing state/results between applications in sequence, and the paper reports that agents largely fail at tasks requiring conditional judgment across >=3 applications, stalling at early sub-goals -- direct evidence that later sub-goals in the chain depend on correctly completing and carrying forward the results of earlier ones.",
  "n_goals": "181 tasks, average of 5.0 sub-goals per task across 17 desktop applications",
  "tracking_demand": "The agent must track progress across a sequence of sub-goals spanning multiple desktop applications, carrying forward state and conditional-judgment results between applications, within a per-difficulty-level step budget (15-40 steps).",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": "max step budget 15 (L1) / 25 (L2) / 40 (L3) / 20 (L4); avg minimum action steps 9.67-27.81 by level",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task, by difficulty level",
  "horizon_stated": "yes",
  "headline_result": "All computer-use agents perform poorly on multi-application tasks (<21% success rate), with the best setting reaching about 20% final success, far below single-app task performance.",
  "availability": "github.com/HITsz-TMG/WindowsWorld",
  "goal_span": "The resulting benchmark contains 181 tasks with an average of 5.0 sub-goals across 17 common desktop applications, of which 78% are inherently multi-application. ... They largely fail at tasks requiring conditional judgment and reasoning across $\\geq$ 3 applications, stalling at early sub-goals",
  "horizon_span": "The average minimum action steps are 9.67 for L1, 18.13 for L2, and 27.81 for L3 ... Each task is executed under a fixed maximum step budget that depends on task level: 15 (L1), 25 (L2), 40 (L3), and 20 (L4).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287916019",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279410951",
  "title": "AgentSynth: Scalable Task Generation for Generalist Computer-Use Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 32,
  "publication_date": "2025-06-17",
  "months_since_pub": 15,
  "citations_per_month": 2.13,
  "artifact_name": "AgentSynth",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete long-horizon computer-use tasks composed of a controllable number of simple subtasks (difficulty levels 1-6); generalize from simple generation-time subtasks that become significantly harder once composed",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Subtasks are individually simple but composed sequentially via information asymmetry, so completing a level-N task requires successfully completing all of its constituent subtasks in order; difficulty is directly modulated by the number of subtasks chained.",
  "n_goals": "over 6,000 tasks; difficulty controlled by subtask count across levels 1-6",
  "tracking_demand": "Agent must carry state correctly across a chain of composed subtasks, since success rate drops steeply as the number of chained subtasks increases from difficulty level 1 to 6.",
  "scoring": "other:success-rate-by-difficulty-level",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "State-of-the-art LLM agents suffer a steep performance drop from 18% success at difficulty level 1 to just 4% at level 6; no human baseline given.",
  "availability": "https://github.com/sunblaze-ucb/AgentSynth",
  "goal_span": "AgentSynth constructs subtasks that are simple during generation but significantly more challenging when composed into long-horizon tasks, enabling the creation of over 6,000 diverse and realistic tasks. A key strength of AgentSynth is its ability to precisely modulate task complexity by varying the number of subtasks. Empirical evaluations show that state-of-the-art LLM agents suffer a steep performance drop, from 18% success at difficulty level 1 to just 4% at level 6",
  "horizon_span": "A key strength of AgentSynth is its ability to precisely modulate task complexity by varying the number of subtasks. Empirical evaluations show that state-of-the-art LLM agents suffer a steep performance drop, from 18% success at difficulty level 1 to just 4% at level 6",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279410951",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284487216",
  "title": "KGCE: Knowledge-Augmented Dual-Graph Evaluator for Cross-Platform Educational Agent Benchmarking with Multimodal Language Models",
  "year": 2025,
  "venue": "IEEE International Conference on Systems, Man and Cybernetics",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 0,
  "publication_date": "2025-10-05",
  "months_since_pub": 11,
  "citations_per_month": 0.0,
  "artifact_name": "KGCE",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a school-specific-software task correctly on Windows, Android, or across both platforms in coordination; satisfy each of the multiple sub-goals a complex task is decomposed into (e.g. one complex task decomposed into five subtasks), each independently verified",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A complex task's sub-goals (e.g. the paper's example of five subtasks for one task) generally must be completed in a structurally-dependent order across platforms/apps, and cross-platform collaborative tasks require actions on one platform to correctly set up state needed by a later action on another.",
  "n_goals": "104 education-related tasks across Windows, Android, and cross-platform scenarios; example complex task decomposed into five subtasks",
  "tracking_demand": "The agent must track completion status of each decomposed sub-goal (via the dual-graph evaluation framework) across potentially multiple platforms and school-specific private-domain software, whose structural specifics are not otherwise well understood by general-purpose agents.",
  "scoring": "subgoal-checkpoint-partial-credit -- the dual-graph evaluation framework decomposes tasks into multiple sub-goals and verifies their completion status, providing fine-grained evaluation metrics, i.e. genuine per-sub-goal credit rather than one aggregate pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/Kinginlife/KGCE",
  "goal_span": "KGCE introduces a dual-graph evaluation framework that decomposes tasks into multiple sub-goals and verifies their completion status, providing fine-grained evaluation metrics.",
  "horizon_span": "we constructed a dataset comprising 104 education-related tasks, covering Windows, Android, and cross-platform collaborative tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284487216",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280045575",
  "title": "MMBench-GUI: Hierarchical Multi-Platform Evaluation Framework for GUI Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 61,
  "publication_date": "2025-07-25",
  "months_since_pub": 14,
  "citations_per_month": 4.36,
  "artifact_name": "MMBench-GUI",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "understand GUI content and ground specific elements accurately (Element Grounding); automate a full task end-to-end within one application/platform (Task Automation); collaborate across multiple tasks/applications requiring cross-platform generalization (Task Collaboration)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The four levels build on each other: correct content understanding is a prerequisite for accurate element grounding, which is a prerequisite for successful task automation, which composes into cross-task/cross-app task collaboration.",
  "n_goals": null,
  "tracking_demand": "Agent requires long-context memory across many actions, tracking of task-planning state, and long-term reasoning to sustain grounded, efficient action sequences without redundant steps across the hierarchy.",
  "scoring": "other:efficiency-quality-area-eqa-plus-hierarchical-levels - proposes a novel Efficiency-Quality Area (EQA) metric assessing GUI agent execution efficiency alongside standard success across four hierarchical levels; this is a milestone-rubric-like scoring across levels, though the abstract does not detail fine per-subtask checkpoint credit within Task Automation/Collaboration.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/open-compass/MMBench-GUI",
  "goal_span": "We introduce MMBench-GUI, a hierarchical benchmark for evaluating GUI automation agents across Windows, macOS, Linux, iOS, Android, and Web platforms. It comprises four levels: GUI Content Understanding, Element Grounding, Task Automation, and Task Collaboration, covering essential skills for GUI agents.",
  "horizon_span": "Furthermore, to achieve reliable GUI automation, an agent requires strong task planning and cross-platform generalization abilities, with long-context memory, a broad action space, and long-term reasoning playing a critical role.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280045575",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "275757225",
  "title": "Mobile-Agent-E: Self-Evolving Mobile Assistant for Complex Tasks",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 134,
  "publication_date": "2025-01-20",
  "months_since_pub": 20,
  "citations_per_month": 6.7,
  "artifact_name": "Mobile-Eval-E",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "break a complex mobile task into subgoals via a Manager agent; execute fine-grained low-level actions per subgoal via an Operator agent; verify and correct action errors via an Action Reflector; aggregate information across steps via a Notetaker; complete long-horizon, multi-app mobile interactions",
  "goal_origin": "mixed:top-level-task-given-up-front-subgoals-self-generated-by-Manager",
  "decomposition": "hierarchical",
  "interdependence": "The Manager's subgoal plan constrains the scope and order of the Operator's low-level actions; the Action Reflector's error verification can trigger plan revision, and the Notetaker's aggregated information must carry forward correctly across multiple apps within the same task.",
  "n_goals": null,
  "tracking_demand": "The system must track the Manager's subgoal plan, the Notetaker's aggregated information, self-evolved long-term Tips/Shortcuts memory, and cross-app interaction state across a single complex mobile task.",
  "scoring": "other:task-completion-with-hierarchical-agent-attribution. No explicit subgoal-checkpoint partial-credit rubric is stated; the headline metric is an aggregate absolute improvement in overall task success rate.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Mobile-Agent-E achieves a 22% absolute improvement over previous state-of-the-art approaches across three foundation model backbones on Mobile-Eval-E; no human/expert baseline reported.",
  "availability": "https://x-plug.github.io/MobileAgent",
  "goal_span": "The framework comprises a Manager, responsible for devising overall plans by breaking down complex tasks into subgoals, and four subordinate agents--Perceptor, Operator, Action Reflector, and Notetaker--which handle fine-grained visual perception, immediate action execution, error verification, and information aggregation, respectively.",
  "horizon_span": "we introduce Mobile-Eval-E, a new benchmark featuring complex mobile tasks requiring long-horizon, multi-app interactions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:275757225",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279260914",
  "title": "What Limits Virtual Agent Application? OmniBench: A Scalable Multi-Dimensional Benchmark for Essential Virtual Agent Capabilities",
  "year": 2025,
  "venue": "International Conference on Machine Learning",
  "tier": "in",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 9,
  "publication_date": "2025-06-10",
  "months_since_pub": 15,
  "citations_per_month": 0.6,
  "artifact_name": "OmniBench / OmniEval",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete a graph-structured task composed of multiple synthesized subtasks of controllable complexity; satisfy subtask-level correctness at each node of the task graph, not just the final outcome; demonstrate each of 10 evaluated virtual-agent capabilities",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are graph-based compositions of subtasks, so completing later subtasks in the graph depends on correctly completing earlier connected subtasks, with an automated pipeline controlling overall task complexity via subtask composition.",
  "n_goals": "36k graph-structured tasks across 20 scenarios",
  "tracking_demand": "Agent must track progress through the task graph's subtask nodes and correctly compose primitive actions across the graph to satisfy subtask-level and graph-based metrics.",
  "scoring": "subgoal-checkpoint-partial-credit - OmniEval includes subtask-level evaluation and graph-based metrics in addition to comprehensive tests across 10 capabilities; explicit subgoal (subtask)-level partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://omni-bench.github.io/",
  "goal_span": "we introduce OmniBench, a self-generating, cross-platform, graph-based benchmark with an automated pipeline for synthesizing tasks of controllable complexity through subtask composition. To evaluate the diverse capabilities of virtual agents on the graph, we further present OmniEval, a multidimensional evaluation framework that includes subtask-level evaluation, graph-based metrics, and comprehensive tests across 10 capabilities.",
  "horizon_span": "Our synthesized dataset contains 36k graph-structured tasks across 20 scenarios, achieving a 91% human acceptance rate.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279260914",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276482111",
  "title": "PC-Agent: A Hierarchical Multi-Agent Collaboration Framework for Complex Task Automation on PC",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 39,
  "publication_date": "2025-02-20",
  "months_since_pub": 19,
  "citations_per_month": 2.05,
  "artifact_name": "PC-Eval",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "decompose complex user instructions into Instruction-Subtask-Action levels; track progress across interdependent subtasks via a Progress agent; make step-by-step decisions via a Decision agent; provide timely bottom-up error feedback and adjustment via a Reflection agent; complete each of 25 real-world complex PC instructions",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Subtasks are interdependent within a decomposed instruction: the Progress agent's tracking of subtask completion informs the Decision agent's next step-by-step choice, and the Reflection agent's bottom-up error feedback can require revising earlier subtask assignments, so an error at one subtask can propagate corrections back up the hierarchy.",
  "n_goals": "25 real-world complex instructions in PC-Eval",
  "tracking_demand": "The multi-agent system must track subtask decomposition and progress (via the Progress agent), current decision state (via the Decision agent), and bottom-up error feedback (via the Reflection agent) across intra- and inter-app workflows on a PC.",
  "scoring": "binary-final-success (task success rate). The abstract reports an overall task success-rate improvement over baselines, with no explicit statement of subgoal-level partial credit despite the Progress agent's explicit tracking role.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "PC-Agent achieves a 32% absolute improvement in task success rate over previous state-of-the-art methods on PC-Eval (25 real-world complex instructions); no human/expert baseline given.",
  "availability": "https://github.com/X-PLUG/MobileAgent/tree/main/PC-Agent",
  "goal_span": "we propose a hierarchical multi-agent collaboration architecture that decomposes decision-making processes into Instruction-Subtask-Action levels. Within this architecture, three agents (i.e., Manager, Progress and Decision) are set up for instruction decomposition, progress tracking and step-by-step decision-making respectively.",
  "horizon_span": "we also introduce a new benchmark PC-Eval with 25 real-world complex instructions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276482111",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280536996",
  "title": "SEAgent: Self-Evolving Computer Use Agent with Autonomous Learning from Experience",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 61,
  "publication_date": "2025-08-06",
  "months_since_pub": 13,
  "citations_per_month": 4.69,
  "artifact_name": "SEAgent (OS-World novel software environments)",
  "artifact_kind": "environment/simulator",
  "domain": "OS-computer-use",
  "goal_types": "autonomously master a novel, unfamiliar software environment through experiential trial-and-error; progressively tackle auto-generated tasks organized from simple to complex (curriculum); assess step-wise trajectory correctness via a World State Model; integrate individual specialist experience into a stronger generalist computer-use agent",
  "goal_origin": "mixed:initial-software-environment-given-up-front-task-curriculum-self-generated",
  "decomposition": "hierarchical",
  "interdependence": "Later curriculum tasks are generated to be increasingly complex based on the agent's demonstrated mastery of simpler auto-generated tasks, so competence gained on earlier easier tasks is a precondition for tackling harder later tasks; the specialist-to-generalist strategy further integrates experience across multiple separately-trained specialists.",
  "n_goals": null,
  "tracking_demand": "The agent must track its step-wise trajectory quality (via the World State Model), its progress along an increasingly difficult auto-generated curriculum, and accumulated experiential insights to be integrated from specialist to generalist training.",
  "scoring": "other:success-rate-improvement-with-curriculum-and-specialist-integration. The abstract reports an aggregate success-rate improvement (11.3% to 34.5%, +23.2 points) on five novel OS-World software environments versus a competitive open-source baseline (UI-TARS), rather than a discrete subgoal-checkpoint scheme, though the curriculum itself implies graded task difficulty.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "SEAgent achieves a 23.2-percentage-point improvement in success rate (from 11.3% to 34.5%) over UI-TARS across five novel OS-World software environments; no human/expert baseline given.",
  "availability": null,
  "goal_span": "SEAgent empowers computer-use agents to autonomously master novel software environments via experiential learning, where agents explore new software, learn through iterative trial-and-error, and progressively tackle auto-generated tasks organized from simple to complex.",
  "horizon_span": "we validate the effectiveness of SEAgent across five novel software environments within OS-World",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280536996",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283226243",
  "title": "UI-CUBE: Enterprise-Grade Computer Use Agent Benchmarking Beyond Task Accuracy to Operational Reliability",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "OS-computer-use",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "OS-computer-use",
  "citation_count": 3,
  "publication_date": "2025-11-21",
  "months_since_pub": 10,
  "citations_per_month": 0.3,
  "artifact_name": "UI-CUBE",
  "artifact_kind": "benchmark",
  "domain": "OS-computer-use",
  "goal_types": "complete simple UI interactions (136 tasks); complete complex copy-paste workflows spanning multiple applications (50 tasks); complete complex enterprise application scenarios (40 tasks); maintain operational reliability (not just functional correctness) under systematic interface variation and multi-resolution testing",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Complex workflows (copy-paste, enterprise scenarios) require chaining multiple simple UI interactions correctly across applications/screens, so failures accumulate: a single misstep in memory, hierarchical planning, or state coordination can break the entire multi-step workflow even when individual UI interactions succeed.",
  "n_goals": "226 tasks total: 136 simple UI interactions, 50 copy-paste tasks, 40 enterprise application scenarios",
  "tracking_demand": "Agent must track state across chained multi-step workflows (e.g., what was copied and where it must be pasted), interface variations, and multiple screen resolutions, verifying success via application-state validation.",
  "scoring": "other:success-rate-by-difficulty-tier - reports success rates separately for simple UI interactions vs. complex workflows (a sharp capability cliff: 67-85% vs. 9-19%), each validated via automated application-state checks; implies per-task-tier binary success rather than in-task subgoal-level partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Simple UI interactions achieve 67-85% success (vs. 97.9% human performance), but complex workflows drop to 9-19% success; human evaluators achieve only 61.2% on complex tasks despite near-perfect performance on simple tasks, establishing realistic performance ceilings.",
  "availability": null,
  "goal_span": "Our evaluation covers simple UI interactions (136 tasks) and complex workflows including copy-paste tasks (50 tasks) and enterprise application scenarios (40 tasks), with systematic interface variation coverage, multi-resolution testing and automated validation of task success through the application state.",
  "horizon_span": "Our evaluation covers simple UI interactions (136 tasks) and complex workflows including copy-paste tasks (50 tasks) and enterprise application scenarios (40 tasks).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283226243",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289132607",
  "title": "Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-06-08",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "50-task hotel expense benchmark (Dynamics 365 F&O)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "itemize every line item of a hotel expense report completely and accurately using enterprise MCP tools; avoid context overflow / stale-state errors while repeatedly calling verbose enterprise tool APIs across the itemization workflow",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Completing later line items depends on maintaining accurate context of already-processed tool responses without stale or overflowing context; verbose tool responses from earlier calls compete for context budget with the information needed to correctly itemize later line items.",
  "n_goals": "50 tasks; five expense types grouped into three categories; four GPT-5 configurations x 5 independent runs each",
  "tracking_demand": "The agent must track already-itemized line items, running token/context budget, and must avoid stale or overflowing tool-call history while repeatedly invoking enterprise MCP tools to reach complete, accurate itemization.",
  "scoring": "continuous-reward -- completion is measured as percent complete itemization (e.g. 8.0% to 91.6% across configurations) and percent amount itemized (e.g. 99.64%), a continuous completeness metric rather than a subgoal-checkpoint rubric; the abstract does not describe discrete per-line-item partial credit beyond these continuous percentages.",
  "horizon_value": "full-context retention: 1,480,996 tokens and 14.56 hours per benchmark (50 tasks x 5 runs); pruning to last 5 tool calls: 535,274 tokens and 5.39 hours; pruning + summarization: 553,374 tokens and 5.79 hours",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "aggregate across the whole 50-task benchmark run (results averaged across 5 independent runs), not a single-task figure",
  "horizon_stated": "yes",
  "headline_result": "Best configuration (pruning + summarization) achieves 91.6% complete itemization and 99.64% average amount itemized; no human/expert baseline is reported, only comparisons across GPT-5 configurations (and cross-model evidence with Claude Sonnet 4.5).",
  "availability": null,
  "goal_span": "We evaluate four GPT-5 configurations on a 50-task hotel expense benchmark: no user model, full conversation history, context pruned to the last 5 tool call/response pairs, and pruning with automated summarization... The no-user-model baseline achieves only 8.0% complete itemization. Full-context retention improves completion to 71.0%... Adding summarization achieves the best result: 91.6% complete itemization and 99.64% average amount itemized.",
  "horizon_span": "Full-context retention improves completion to 71.0%, but consumes 1,480,996 tokens and 14.56 hours per benchmark. Pruning to the last 5 tool calls improves completion to 79.0% while reducing token use to 535,274 and runtime to 5.39 hours. Adding summarization achieves the best result: 91.6% complete itemization and 99.64% average amount itemized, with 553,374 tokens and 5.79 hours.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289132607",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285615092",
  "title": "AD-Bench: A Real-World, Trajectory-Aware Advertising Analytics Benchmark for LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 3,
  "publication_date": "2026-02-15",
  "months_since_pub": 7,
  "citations_per_month": 0.43,
  "artifact_name": "AD-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "answer a real user marketing-analysis request via multi-round, multi-tool collaboration; maintain trajectory coverage consistent with an expert tool-call trajectory; produce answers that remain correct/consistent with a continuously evolving production advertising platform (dynamic ground truth); succeed across three stratified difficulty levels (L1-L3) requiring increasing multi-round, multi-tool collaboration",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Because ground-truth answers are regenerated dynamically by replaying expert tool-call trajectories against the current environment, an agent's answer depends on correctly executing an analogous multi-tool sequence itself, and difficulty levels (L1-L3) escalate the number of coordinated tool/round dependencies required.",
  "n_goals": "three difficulty levels (L1-L3) stratifying multi-round, multi-tool collaboration",
  "tracking_demand": "The agent must track which professional tools it has called and in what order across multiple rounds, since evaluation jointly measures end-to-end answer correctness (Pass@k) and how well its trajectory covers the expert reference trajectory, especially as this compounds at higher difficulty levels.",
  "scoring": "other:pass-at-k-plus-trajectory-coverage",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The best model, Claude-Opus-4.7, attains Pass@1=76.9% and Pass@3=80.4% with 82.7% trajectory coverage overall, but drops sharply on the hardest tier (L3) to Pass@1=61.4%/Pass@3=65.1%; no explicit human/expert baseline is given (comparison is model-vs-model across difficulty tiers).",
  "availability": null,
  "goal_span": "a trajectory-aware evaluation that jointly measures end-to-end answer correctness (Pass@k) and trajectory coverage. Requests are stratified into three difficulty levels (L1-L3) to probe multi-round, multi-tool collaboration.",
  "horizon_span": "Requests are stratified into three difficulty levels (L1-L3) to probe multi-round, multi-tool collaboration.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285615092",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284909788",
  "title": "APEX-Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 20,
  "publication_date": "2026-01-20",
  "months_since_pub": 8,
  "citations_per_month": 2.5,
  "artifact_name": "APEX-Agents",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "execute a long-horizon, cross-application task created by investment-banking analysts, management consultants, or corporate lawyers; navigate realistic work environments containing files and tools to produce the required deliverable",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Cross-application tasks require moving between multiple files/tools within one realistic work environment, so completing later parts of a task depends on outputs or state produced while using earlier files/tools in the same task.",
  "n_goals": "480 tasks (n=480)",
  "tracking_demand": "Agent must track which files and tool states it has already produced/consulted while navigating a cross-application work environment, using rubrics and gold outputs to determine success (Pass@1).",
  "scoring": "other:not-stated precisely \u2014 the abstract reports Pass@1 leaderboard scores per model but does not describe explicit subgoal-checkpoint partial credit within a task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Gemini 3 Flash (Thinking=High) achieves the highest score of 24.0% Pass@1, followed by GPT-5.2, Claude Opus 4.5, and Gemini 3 Pro (Thinking=High) (no human baseline given).",
  "availability": "APEX-Agents benchmark (n=480) with prompts, rubrics, gold outputs, files, and metadata is open-sourced, along with the Archipelago execution/evaluation infrastructure; no specific URL given in the abstract.",
  "goal_span": "APEX-Agents requires agents to navigate realistic work environments with files and tools. We test eight agents for the leaderboard using Pass@1.",
  "horizon_span": "APEX-Agents requires agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284909788",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "291182992",
  "title": "AeroCopilotBench: A Two-Tier Benchmark for Evaluating LLM Agents as Aviation Copilots in an Interactive Virtual Cockpit Environment",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:aviation-procedural-work",
  "citation_count": 0,
  "publication_date": "2026-08-17",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "AeroCopilotBench (ACOE)",
  "artifact_kind": "benchmark",
  "domain": "other:aviation-procedural-work",
  "goal_types": "diagnose faults and operate aircraft systems through standardized tool interfaces during emergency/abnormal procedures; achieve all task goal-conditions for each of 73 Tier-2 emergency/abnormal tasks; avoid violating any hard safety constraint while progressing toward goals",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A trajectory succeeds only when all task goal-conditions are achieved without violating any hard safety constraint, so goal progress and safety-constraint compliance must be jointly tracked and satisfied throughout the cockpit's state transitions.",
  "n_goals": "73 emergency/abnormal Tier-2 tasks (plus 1,200 Tier-1 multiple-choice knowledge questions)",
  "tracking_demand": "Agent must interpret cockpit state, diagnose faults, and operate aircraft systems while simultaneously tracking goal-condition progress and hard safety-constraint compliance across the executable procedure.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Across 12 models, the highest Tier-2 success rate is 72.6%, while static knowledge performance does not consistently translate into procedural execution; no human/expert baseline given.",
  "availability": null,
  "goal_span": "We establish a safety-gated evaluation framework in which a trajectory succeeds only when all task goals are achieved without violating any hard safety constraint, while safe goal progress and trajectory safety are measured separately.",
  "horizon_span": "Tier-2 comprises 73 emergency and abnormal tasks derived from the manufacturers'Pilot's Operating Handbooks (POHs) and instantiated in ACOE.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291182992",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290377535",
  "title": "Agentic ERP: Multi-Agent Large Language Model Architecture for Autonomous Enterprise Resource Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-19",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "Agentic ERP",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "execute end-to-end business workflows across ERP functional boundaries (e.g. procurement, inventory, crisis response); resolve cross-functional crisis tasks via a Planner-Executor-Reflector-Responder orchestration; sustain a full simulated year of ERP operation while avoiding stockouts, compared against rule-based RPA and no-intervention baselines",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "Autonomous ERP operation is formulated as a constrained sequential-decision problem over a structured enterprise state, where role-aligned agents' actions share and update the same enterprise state, so an earlier decision (e.g. under-ordering stock) constrains or creates the crisis conditions later agents must resolve.",
  "n_goals": "a scenario-based task suite; a comparison of six orchestration paradigms on cross-functional crisis tasks; a 365-day agent-in-the-loop simulation",
  "tracking_demand": "The agent(s) must track the structured enterprise state (inventory, demand stream, crisis conditions) across a full simulated year, coordinate across role-aligned agents via externalised grading criteria and sprint contracts, and avoid accumulating stockouts as the rule-based baseline does.",
  "scoring": "other:comparative-outcome-metrics -- performance is compared across a scenario-based task suite, an orchestration-paradigm comparison, and a full-year simulation against rule-based RPA and no-intervention baselines; the abstract does not describe a fine-grained subgoal-checkpoint rubric beyond these comparative outcome metrics.",
  "horizon_value": "a 365-day agent-in-the-loop simulation (a simulated year of ERP operation)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/run (one full simulated-year ERP operation trajectory)",
  "horizon_stated": "yes",
  "headline_result": "The proposed multi-agent method sustains a simulated year of operation with zero stockouts, while the rule-based RPA baseline accumulates hundreds under the same demand stream; no human-expert baseline comparison is reported.",
  "availability": null,
  "goal_span": "the system is evaluated at three levels: a scenario-based task suite, a comprehensive comparison of six orchestration paradigms on cross-functional crisis tasks, and a 365-day agent-in-the-loop simulation against rule-based RPA and no-intervention baselines. Across these levels the proposed multi-agent method is significantly better than the baseline, and the system sustains a simulated year of operation with zero stockouts while the rule-based baseline accumulates hundreds under the same demand stream.",
  "horizon_span": "the system is evaluated at three levels: a scenario-based task suite, a comprehensive comparison of six orchestration paradigms on cross-functional crisis tasks, and a 365-day agent-in-the-loop simulation against rule-based RPA and no-intervention baselines.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290377535",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285303367",
  "title": "AgenticPay: A Multi-Agent LLM Negotiation System for Buyer-Seller Transactions",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 13,
  "publication_date": "2026-02-05",
  "months_since_pub": 7,
  "citations_per_month": 1.86,
  "artifact_name": "AgenticPay",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "negotiate to reach agreement given private constraints and product-dependent valuations; maximize feasibility, efficiency, and welfare across a negotiation; successfully complete each of 110+ tasks ranging from bilateral bargaining to many-to-many markets",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Within a negotiation, each party's private constraints and valuations plus multi-round natural-language offers/counteroffers constrain what agreement is reachable; in many-to-many markets, one pair's agreement can affect resources or prices available to other pairs.",
  "n_goals": "over 110 tasks ranging from bilateral bargaining to many-to-many markets",
  "tracking_demand": "Agents must track their own private constraints/valuations, the evolving history of natural-language offers and counteroffers across negotiation rounds, and, in many-to-many markets, the state of other concurrent negotiations.",
  "scoring": "other:feasibility-efficiency-and-welfare-metrics. The framework reports structured metrics for feasibility, efficiency, and welfare rather than a single binary success score; the abstract does not describe explicit per-round subgoal-checkpoint credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Benchmarking state-of-the-art proprietary and open-weight LLMs reveals substantial gaps in negotiation performance and long-horizon strategic reasoning challenges; no specific top model/number named in the abstract.",
  "availability": "https://github.com/SafeRL-Lab/AgenticPay",
  "goal_span": "AgenticPay models markets in which buyers and sellers possess private constraints and product-dependent valuations, and must reach agreements through multi-round linguistic negotiation rather than numeric bidding alone. The framework supports a diverse suite of over 110 tasks ranging from bilateral bargaining to many-to-many markets, with structured action extraction and metrics for feasibility, efficiency, and welfare.",
  "horizon_span": "must reach agreements through multi-round linguistic negotiation rather than numeric bidding alone",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285303367",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288701037",
  "title": "AgenticVBench: Can AI Agents Complete Real-World Post-Production Tasks?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:creative-post-production-workflow",
  "citation_count": 3,
  "publication_date": "2026-05-26",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "AgenticVBench",
  "artifact_kind": "benchmark",
  "domain": "other:creative-post-production-workflow",
  "goal_types": "complete each of 100 agentic post-production tasks across 4 task families; compose capabilities across text, image, audio, and video understanding within one task; plan and execute long-horizon multi-step production workflows using appropriate tools",
  "goal_origin": "given-up-front",
  "decomposition": "other:four-task-families-each-with-real-production-workflow-structure",
  "interdependence": "Post-production tasks require composite, cross-modal capabilities (text/image/audio/video) plus tool use combined within a single long-horizon workflow, so a failure in one modality-specific step can block downstream video-production steps that depend on its output.",
  "n_goals": "100 agentic tasks across 4 task families",
  "tracking_demand": "The agent must track progress through a long-horizon, multi-modal production workflow and correctly sequence tool use, since tasks are constructed from real production workflows contributed by industry experts and evaluated jointly by programmatic verifiers and expert rubrics.",
  "scoring": "other:programmatic-verifiers-plus-expert-rubrics",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The best evaluated agent stack barely crosses 30%, far below human expert performance on the same tasks (exact human score not given in the abstract); harness choice substantially affects scores, tool-use patterns, and failure modes.",
  "availability": "https://agenticvbench.com",
  "goal_span": "Tasks are paired with evaluation specifications that combine programmatic verifiers and expert rubrics.",
  "horizon_span": "they require composite capabilities across text, image, audio, and video understanding, along with long-horizon planning, and tool use",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288701037",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288977338",
  "title": "Agents' Last Exam",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 12,
  "publication_date": "2026-06-03",
  "months_since_pub": 3,
  "citations_per_month": 4.0,
  "artifact_name": "Agents' Last Exam (ALE)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete a long-horizon, economically valuable, real-world professional task with a verifiable outcome, drawn from a specific occupational sub-field; achieve sustained performance across the task rather than a single-shot correct answer, particularly on the hardest ('last-exam') tier",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Not detailed at the cross-task level (tasks across the 55 sub-fields/13 clusters are largely independent occupational tasks); within a single long-horizon task, the abstract does not specify explicit sub-goal interdependence beyond 'real world' and 'verifiable' framing; none stated.",
  "n_goals": "1K+ tasks organized into 55 sub-fields grouped into 13 industry clusters, developed with 250+ industry experts; designed as a continuously growing 'living benchmark'",
  "tracking_demand": "The agent must sustain performance on long-horizon, economically valuable real-world tasks whose outcomes are verifiable, across an occupational taxonomy spanning 55 sub-fields and 13 industry clusters, with the hardest tier proving far from saturated (average full pass rate below 1%).",
  "scoring": "binary-final-success -- performance is reported as a 'full pass rate' per tier (below 1% average on the hardest tier, 2.6% for frontier agents, 26% overall across all tiers), consistent with an all-or-nothing full-task-pass metric rather than a stated subgoal-checkpoint partial-credit rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier agents average a 2.6% pass rate on the hardest 'last-exam' tier (26% overall across all task tiers); average full pass rate across mainstream harness/backbone configurations is below 1% on the hardest tier. No human/expert baseline comparison is given in the retrieved abstract.",
  "availability": null,
  "goal_span": "This paper introduces Agents' Last Exam (ALE), a benchmark designed to evaluate AI agents on long horizon, economically valuable, real world tasks with verifiable outcomes. Developed in collaboration with 250+ industry experts, ALE covers non-physical industries defined with reference to O*NET / SOC 2018 (the U.S. federal occupational taxonomy). It is organized around a task taxonomy with 55 sub fields grouped into 13 industry clusters covering 1K+ tasks. Current results show that the hardest tier remains far from saturated: across mainstream harness and backbone configurations, the average full pass rate is below 1%.",
  "horizon_span": "It is organized around a task taxonomy with 55 sub fields grouped into 13 industry clusters covering 1K+ tasks. Current results show that the hardest tier remains far from saturated: across mainstream harness and backbone configurations, the average full pass rate is below 1%.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288977338",
  "provenance": "web-registry",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287636396",
  "title": "AutomationBench",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-04-21",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "AutomationBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "discover the relevant REST API endpoints needed for a cross-application workflow (CRM, inbox, calendar, messaging, etc.); follow layered business-policy rules while writing data; navigate environments containing irrelevant or misleading records without being derailed; get correct data into the right systems by the end of the workflow (end-state correctness)",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A single task 'may span a CRM, inbox, calendar, and messaging platform', so endpoint discovery in one system must be correctly sequenced with policy-compliant writes to others, and getting one system's state wrong can prevent the correct end-state from being reached across the whole cross-application workflow.",
  "n_goals": null,
  "tracking_demand": "Agent must track which endpoints it has discovered across multiple applications, which business-policy rules apply to each write, and whether records encountered are relevant or misleading, to ensure the correct final data lands in each system.",
  "scoring": "binary-final-success \u2014 'Grading is programmatic and end-state only: whether the correct data ended up in the right systems', with the abstract explicitly noting there is no partial/intermediate credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the best frontier models currently score below 10% (no human baseline given).",
  "availability": null,
  "goal_span": "Grading is programmatic and end-state only: whether the correct data ended up in the right systems.",
  "horizon_span": "tasks span Sales, Marketing, Operations, Support, Finance, and HR domains",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287636396",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288903642",
  "title": "BigFinanceBench: A Workflow-Grounded Benchmark for Financial-Research Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 5,
  "publication_date": "2026-06-02",
  "months_since_pub": 3,
  "citations_per_month": 1.67,
  "artifact_name": "BigFinanceBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "produce an auditable financial-research derivation (source choice, period/accounting definition, assumptions, calculation), not just a final answer; satisfy each of many independently checkable rubric points per item (36,241 points across 928 items); correctly complete each of 928 expert-authored open-ended financial-research tasks",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each rubric point checks one independently-checkable step of the derivation (source, period, definition, assumption, or calculation), and final-answer correctness is shown to be a lossy proxy for whether the full derivation is right, so getting the final number right does not guarantee all constituent steps were done correctly.",
  "n_goals": "928 items; 36,241 rubric points total (about 39 points/item on average)",
  "tracking_demand": "The agent must track and justify each auditable component of its derivation (data source, period/accounting definition, assumptions, calculation steps) so that an independent rubric can check each step, not just the final numeric answer.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The best of ten evaluated frontier/open-weight agents reaches only 58.8% rubric score; final-answer accuracy is shown to be a lossy proxy for derivation quality; no explicit human/expert-analyst score is given as a comparison point.",
  "availability": null,
  "goal_span": "We introduce BigFinanceBench, a 928-item expert-authored benchmark of open-ended financial-research tasks in which each item pairs a ground-truth reference answer with a point-weighted rubric that decomposes the derivation into independently checkable steps... Across 36,241 rubric points, the benchmark supports partial-credit evaluation and localization of failures across the analyst workflow.",
  "horizon_span": "which source was chosen, which period and accounting definition were used, which assumptions were made, and how the calculation was performed",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288903642",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288814344",
  "title": "BlueFin: Benchmarking LLM Agents on Financial Spreadsheets",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-05-29",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "BlueFin",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "synthesize new spreadsheet content/formulas from source data; manipulate existing financial spreadsheet workbooks (edits, restructuring); comprehend and answer questions about spreadsheet content; satisfy each of up to 3,225 granular rubric criteria across 131 tasks",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Rubric criteria within a task jointly define whether the completed workbook is correct; the abstract notes weaknesses in 'dynamic correctness', implying later formula/state correctness depends on earlier manipulation steps within the same workbook.",
  "n_goals": "131 tasks containing 3,225 granular rubric criteria (~24.6 criteria per task on average)",
  "tracking_demand": "The agent must track and satisfy many granular, LM-judge-scored rubric criteria across synthesis, manipulation, and comprehension actions within one workbook, maintaining correctness as the workbook's formulas/state evolve ('dynamic correctness').",
  "scoring": "LLM-judge-rubric -- 3,225 granular rubric criteria are scored by an LM judge validated against expert human annotators (parity alpha=0.826, macro-F1=0.839); this is explicit subgoal/criterion-level partial credit, not a single binary pass/fail.",
  "horizon_value": "tasks require at least 45+ minutes of work for a human analyst to complete from scratch",
  "horizon_unit": "human-expert-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Strongest LLMs achieve less than 50% average scores across tasks, with particular weaknesses in dynamic correctness; no human-expert score is reported (only that tasks require 45+ minutes of human-analyst time).",
  "availability": null,
  "goal_span": "we curate a set of 131 challenging, complex tasks with real-world relevance in the domain, containing 3,225 granular rubric criteria; notably, our rubric criteria and LM judge evaluations are validated by a team of expert human annotators.",
  "horizon_span": "at least 45+ minutes of work for an analyst to complete from scratch",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288814344",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "human-expert-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290989893",
  "title": "Business Arena: Benchmarking LLM Agents in a Realistic Marketplace",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-09",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Business Arena",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "infer business opportunities from partial market signals; commit capital under uncertainty when buying from suppliers; set/adjust pricing to sell to buyers profitably; satisfy regulatory obligations before trading legally; sustain profitable operation of a cross-border shop over a long horizon",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Sourcing, pricing, and recovery decisions are coupled through shared capital and delayed market feedback: committing capital to one supplier purchase constrains what can be spent on pricing/marketing, and outcomes are realized only with delay, so early decisions shape which later opportunities remain viable.",
  "n_goals": null,
  "tracking_demand": "The agent must track capital, delayed and coupled consequences of sourcing/pricing/recovery decisions, market conditions calibrated from real data, and regulatory obligations across a long-horizon cross-border trading operation.",
  "scoring": "other:profit-plus-skill-level-and-action-attribution. Final net worth/profit is the top-line outcome, but the paper also computes skill-level metrics and action-level attribution (tracing gains/losses to specific sourcing/pricing/recovery decisions), a decomposed, non-binary form of credit beyond final profit alone.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "15 frontier models show a ninefold difference in mean final net worth; even the best model falls behind human-designed strategies; no single top model/number is named in the abstract.",
  "availability": null,
  "goal_span": "We introduce Business Arena, a controlled environment where an AI agent runs a cross-border shop, buying from suppliers and selling to buyers over a long horizon. ... action-level attribution identifies the sourcing, pricing, and recovery decisions that create or destroy value.",
  "horizon_span": "an AI agent runs a cross-border shop, buying from suppliers and selling to buyers over a long horizon",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290989893",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289621380",
  "title": "CEO-Bench: Can Agents Play the Long Game?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "open-world-game",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 3,
  "publication_date": "2026-06-16",
  "months_since_pub": 3,
  "citations_per_month": 1.0,
  "artifact_name": "CEO-Bench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "operate a startup for 500 days, managing pricing, marketing, budgeting, and other business aspects; acquire information from noisy, interconnected business databases and translate it into strategy; adapt strategy to a changing world/market over the run; orchestrate many interdependent business decisions toward a coherent long-term financial goal",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Pricing, marketing, and budgeting decisions jointly determine cash flow, churn, and customer losses, so a decision in one area (e.g., aggressive marketing spend) constrains what remains feasible in another (e.g., operating budget) over the 500-day run.",
  "n_goals": null,
  "tracking_demand": "The agent must track noisy, interconnected business databases (churn regimes, billing timing, customer losses, projected cash) and coordinate many interdependent decisions via code across a 500-day simulated run.",
  "scoring": "continuous-reward. Performance is measured by final account balance relative to the $1M starting balance and a rule-based baseline; the abstract describes no discrete subgoal-checkpoint credit scheme, only continuous financial-outcome tracking.",
  "horizon_value": "500",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the entire startup-operation run)",
  "horizon_stated": "yes",
  "headline_result": "Only Claude Fable 5, GPT-5.6 Sol, and Claude Opus 4.8 finish above the $1M starting balance; all evaluated models remain below the rule-based baseline; no external human-CEO baseline given.",
  "availability": null,
  "goal_span": "We introduce CEO-Bench, which evaluates these capabilities together by simulating a representative real-world task: operating a startup for 500 days. An agent manages pricing, marketing, budgeting, and many other aspects of a fictional company through a programmable Python interface.",
  "horizon_span": "simulating a representative real-world task: operating a startup for 500 days",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289621380",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289621702",
  "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-06-16",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "CEO-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "integrate conflicting recommendations from four role-conditioned C-suite advisors (CFO, CTO, COO, CMO) into one allocation plan; redirect capital across business units under information asymmetry and organizational constraints; synthesize advice consistently across multiple rounds with temporal dependencies",
  "goal_origin": "mixed:given-up-front-scenario-with-conflicting-advice-injected-each-round",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each round's allocation plan must reconcile four advisors' private-signal-based conflicting recommendations, and history-sensitive judgment scoring evaluates whether later-round decisions remain consistent with earlier ones.",
  "n_goals": "13 scenarios; four role-conditioned advisors (CFO, CTO, COO, CMO) per scenario",
  "tracking_demand": "Agent (as CEO) must track and reconcile conflicting, privately-signaled recommendations from four C-suite advisors across a multi-round resource-reallocation process, remembering prior decisions for history-sensitive judgment.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "All five evaluated frontier models achieve high structural (plan) validity but diverge sharply on strategic calibration, the hardest capability layer; systematic failure modes include single-advisor capture, conservative default under ambiguity, and historical amnesia; no human baseline given.",
  "availability": null,
  "goal_span": "LLM agents receive conflicting advice from four role-conditioned C-suite advisors (CFO, CTO, COO, CMO), each with private signals and distinct priorities, and must synthesize these into a concrete allocation plan evaluated along four dimensions: role integration, conditional boldness, history-sensitive judgment, and plan validity.",
  "horizon_span": "the process of redirecting capital across business units in a multi-round, constraint-rich organizational environment... Experiments across five frontier models on 13 scenarios reveal that all models achieve high structural validity but diverge sharply on strategic calibration",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289621702",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288650157",
  "title": "CHI-Bench: Can AI Agents Automate End-to-End, Long-Horizon, Policy-Rich Healthcare Workflows?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 6,
  "publication_date": "2026-05-15",
  "months_since_pub": 4,
  "citations_per_month": 1.5,
  "artifact_name": "CHI-Bench (\u03c7-Bench)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "ground each decision in a large library of medical, insurance, and operational policy rules; play multiple roles within a single task, handing off between them; conduct multilateral multi-turn dialogs (e.g. peer-to-peer review, patient outreach) as intermediate workflow steps; drive a clinical case to a terminal status via tool calls and role-artifact writing",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A single task requires the agent to play multiple roles with handoffs, so a later role's actions depend on artifacts/decisions produced by an earlier role, and intermediate multi-turn dialogs (peer review, patient outreach) must resolve before the case can proceed toward a terminal status; performance collapses specifically when all of this is compressed into a single session.",
  "n_goals": "long-horizon workflows across 3 domains, driven via 87 MCP tools over 20 healthcare apps, guided by a 1,290+ document handbook skill",
  "tracking_demand": "The agent must track its current role and handoffs to other roles, policy compliance against a 1,290+ document handbook, and the clinical case's evolving status across a high-fidelity simulator of 20 apps, until it reaches a terminal status.",
  "scoring": "other:strict-pass-rate-across-harness-model-configurations",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Across 30 agent harness/model configurations, the best agent resolves only 28.0% of tasks; no agent clears 20% on strict pass^3; executing all tasks in a single session drops performance to 3.8%; no human baseline given.",
  "availability": "https://github.com/actava-ai/chi-bench",
  "goal_span": "we introduce $\\chi$-Bench, a benchmark of long-horizon healthcare workflows across three domains: provider prior authorization, payer utilization management, and care management. Each task hands the agent a clinical case in a high-fidelity simulator of 20 healthcare apps exposed via 87 MCP tools, which it must drive to a terminal status through tool calls and writing the role's artifacts, guided by a 1,290+ document managed-care operations handbook skill.",
  "horizon_span": "executing all tasks in a single session slumps the performance to 3.8%",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288650157",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287915622",
  "title": "Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 13,
  "publication_date": "2026-04-30",
  "months_since_pub": 5,
  "citations_per_month": 2.6,
  "artifact_name": "Claw-Eval-Live",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete end-to-end units of work across software tools, business services, and local workspaces per release; pass controlled tasks with fixed fixtures/services/workspaces/graders reflecting current public workflow-demand signals; handle HR, management, and multi-system business workflows as well as local workspace repair",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each release's refreshable signal layer is updated from public demand while the task snapshot stays fixed and reproducible; grading combines execution traces, audit logs, service state, and post-run workspace artifacts, so correctness on one task depends on state changes being consistently reflected across all of these evidence sources.",
  "n_goals": "105 tasks per release (ClawHub Top-500 skills used in the current release)",
  "tracking_demand": "Agent must produce verifiable execution traces, audit logs, and consistent service/workspace state across each end-to-end task, since grading uses deterministic checks when evidence is sufficient and structured LLM judging only for semantic dimensions.",
  "scoring": "other:deterministic-checks-plus-structured-llm-judging",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The leading model passes only 66.7% of tasks and no model reaches 70%; HR, management, and multi-system business workflows are persistent bottlenecks, while local workspace repair is comparatively easier but unsaturated; no human baseline given.",
  "availability": null,
  "goal_span": "For grading, Claw-Eval-Live records execution traces, audit logs, service state, and post-run workspace artifacts, using deterministic checks when evidence is sufficient and structured LLM judging only for semantic dimensions. The release contains 105 tasks spanning controlled business services and local workspace repair, and evaluates 13 frontier models under a shared public pass rule.",
  "horizon_span": "The release contains 105 tasks spanning controlled business services and local workspace repair, and evaluates 13 frontier models under a shared public pass rule.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287915622",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287782807",
  "title": "ClawMark: A Living-World Benchmark for Multi-Turn, Multi-Day, Multimodal Coworker Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 15,
  "publication_date": "2026-04-26",
  "months_since_pub": 5,
  "citations_per_month": 3.0,
  "artifact_name": "ClawMark",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "carry a professional coworker task forward correctly across multiple working days; detect and adapt to exogenous environment updates (new emails, calendar shifts, KB edits) that occur between turns; satisfy each of the task's deterministic Python checkers over post-execution service state (mean 15.4 checkers/task); coordinate consistent state across five stateful sandboxed services (filesystem, email, calendar, knowledge base, spreadsheet)",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "The environment evolves independently between turns (new emails, calendar shifts, KB updates), so the agent's plan from an earlier turn may be invalidated by exogenous changes it must detect and adapt to before continuing; the paper reports performance drops specifically after the first such exogenous update.",
  "n_goals": "100 tasks across 13 professional scenarios; mean 3.6 turns (2-6 range) and mean 15.4 checkers (6-29 range) per task",
  "tracking_demand": "The agent must track evolving state across five sandboxed services (filesystem, email, calendar, knowledge base, spreadsheet) over multiple working days, detecting and incorporating exogenous updates injected between turns, verified by up to 29 deterministic checkers per task.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": "2-6 turns per task (mean 3.6); one turn = one in-universe working day",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Strongest model reaches 75.8 weighted score but best strict Task Success is only 20.0%, showing partial progress is common while complete workflow completion remains rare.",
  "availability": null,
  "goal_span": "The current release contains 100 tasks across 13 professional scenarios, executed against five stateful sandboxed services (filesystem, email, calendar, knowledge base, spreadsheet) and scored by 1537 deterministic Python checkers over post-execution service state; no LLM-as-judge is invoked during scoring. ... The strongest model reaches 75.8 weighted score, but the best strict Task Success is only 20.0%",
  "horizon_span": "Tasks range from two to six turns (mean 3.6) and 6 to 29 checkers (mean 15.4).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287782807",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287208605",
  "title": "ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 25,
  "publication_date": "2026-04-06",
  "months_since_pub": 5,
  "citations_per_month": 5.0,
  "artifact_name": "ClawsBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete single-service productivity tasks (e.g. Gmail, Slack, Calendar, Docs, Drive) correctly and safely; complete cross-service workflows spanning multiple mock services while maintaining consistent state; avoid unsafe/irreversible actions in safety-critical scenarios",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Actions across services share persistent state (via full state management with deterministic snapshot/restore), so cross-service tasks require actions in one service to remain consistent with state changes in another, and unsafe/irreversible actions can permanently corrupt that shared state.",
  "n_goals": "44 structured tasks spanning single-service, cross-service, and safety-critical categories",
  "tracking_demand": "Agents must track persistent state across five mock services (Gmail, Slack, Calendar, Docs, Drive), recognize safety-critical constraints to avoid irreversible/unsafe actions, and coordinate behavior across services via a meta-prompt layer.",
  "scoring": "other:dual-axis -- both task success rate (39-64%) and unsafe action rate (7-33%) are reported as separate, non-conflated metrics rather than a single subgoal-checkpoint rubric; the abstract does not describe partial credit within a single task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "With full scaffolding, agents achieve task success rates of 39-64% (top five models on OpenClaw within a 10-point band of 53-63%) but unsafe action rates of 7-33%; no human/expert baseline is reported.",
  "availability": "https://clawsbench.com",
  "goal_span": "It includes five high-fidelity mock services (Gmail, Slack, Google Calendar, Google Docs, Google Drive) with full state management and deterministic snapshot/restore, along with 44 structured tasks covering single-service, cross-service, and safety-critical scenarios.",
  "horizon_span": "Experiments across 6 models, 4 agent harnesses, and 33 conditions show that with full scaffolding, agents achieve task success rates of 39-64% but exhibit unsafe action rates of 7-33%.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287208605",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289300619",
  "title": "CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 6,
  "publication_date": "2026-06-15",
  "months_since_pub": 3,
  "citations_per_month": 2.0,
  "artifact_name": "CoffeeBench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "maximize cumulative net income for one's own firm (coffee roaster) over a long simulated run; manage cash, inventory, and pricing on an ongoing basis; communicate and transact with other heterogeneous firms (farmers, roasters, retailers) to secure supply/demand",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Firms are heterogeneous and interdependent (farmers supply roasters, roasters supply retailers), and communication/transactions between them affect each other's cash, inventory, and pricing decisions, so the evaluated coffee roaster's outcomes over the 90-day run depend on its ongoing relationships with the other five fixed-reference firms.",
  "n_goals": null,
  "tracking_demand": "The agent must track its own cash, inventory, and pricing state day by day over the 90-day simulation, plus the state of its ongoing communications/transactions with other firms, to maximize cumulative net income rather than any single day's outcome.",
  "scoring": "continuous-reward. Firms are evaluated by cumulative net income accumulated over the 90-day simulation; the abstract compares agents against a passive no-action baseline and reports behavioral differences rather than a subgoal-checkpoint rubric.",
  "horizon_value": "90-day simulation",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole 90-day simulated economy run)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "two farmers, two roasters, and two retailers autonomously operate their businesses over a 90-day simulation, each seeking to maximize cumulative net income through communication and transactions while managing cash, inventory, and pricing. The evaluated model controls one coffee roaster, while the remaining firms are controlled by fixed reference agents.",
  "horizon_span": "two farmers, two roasters, and two retailers autonomously operate their businesses over a 90-day simulation",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289300619",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290394586",
  "title": "Do AI-Native Biotechs Need Departments? Benchmarking Company World Models for AI-Driven Drug Development",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-21",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "Company World Model dry-lab benchmark (45 retrospective cases)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "maintain and update a persistent asset-to-value state (Live Asset Value Record) across scientific, regulatory, BD, commercial, financial, and execution constraints; resolve 45 retrospective public-information decision cases with hidden outcomes under strict time cutoffs; achieve success via external BD deals, regulatory approval/launch, and revenue discipline",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Deal, Approval, Revenue, and Investment Arbiter loops all read and update a shared Live Asset Value Record, so a decision in one loop changes the state the other loops must react to.",
  "n_goals": "45 retrospective decision cases",
  "tracking_demand": "The architecture must maintain a persistent, continuously updated asset-to-value state record across scientific, regulatory, BD, commercial, financial, and execution constraints via Deal/Approval/Revenue/Investment Arbiter loops.",
  "scoring": "other:automatic-value-conversion-score-plus-blinded-pairwise-judging",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The value-conversion architecture achieved the highest automatic value-conversion score and was strongly preferred by value-specific blinded judges over human-org-mimic baselines, though a stronger human baseline remained competitive under stress tests (no single named human/model score given).",
  "availability": null,
  "goal_span": "The benchmark contains 45 retrospective public-information decision cases with strict time cutoffs, hidden outcomes, common schemas, automatic scoring, and blinded pairwise judging... The value-conversion architecture is a prompt-level approximation of a Company World Model: a Live Asset Value Record updated by Deal, Approval, Revenue, and Investment Arbiter loops.",
  "horizon_span": "The benchmark contains 45 retrospective public-information decision cases with strict time cutoffs, hidden outcomes, common schemas, automatic scoring, and blinded pairwise judging.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290394586",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285726497",
  "title": "EnterpriseBench Corecraft: Training Generalizable Agents on High-Fidelity RL Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 7,
  "publication_date": "2026-02-18",
  "months_since_pub": 7,
  "citations_per_month": 1.0,
  "artifact_name": "CoreCraft (EnterpriseBench)",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "perform multi-step, domain-specific customer-support work across a simulated enterprise with 2,500+ entities across 14 entity types; satisfy all expert-authored rubric criteria for a given task using 23 available tools",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks act on a shared pool of 2,500+ enterprise entities via 23 tools, so satisfying one rubric criterion (e.g., updating a customer record) can affect state relevant to other criteria for the same task.",
  "n_goals": "expert-authored rubric criteria per task (exact per-task count not given); frontier models solve fewer than 30% of tasks when all criteria must be satisfied",
  "tracking_demand": "Agent must track entity state across the enterprise simulation and verify each expert-authored rubric criterion is satisfied before the task counts as solved.",
  "scoring": "subgoal-checkpoint-partial-credit - task success is defined by satisfying all expert-authored rubric criteria jointly, implying rubric criteria are individually checkable subgoals even though the reported pass rate is all-or-nothing per task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Frontier models such as GPT-5.2 and Claude Opus 4.6 solve fewer than 30% of tasks when all expert-authored rubric criteria must be satisfied; after GRPO training, GLM 4.6 improves from 25.37% to 36.76% task pass rate; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Frontier models such as GPT-5.2 and Claude Opus 4.6 solve fewer than 30% of tasks when all expert-authored rubric criteria must be satisfied.",
  "horizon_span": "CoreCraft is a fully operational enterprise simulation of a customer support organization, comprising over 2,500 entities across 14 entity types with 23 unique tools, designed to measure whether AI agents can perform the multi-step, domain-specific work that real jobs demand.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285726497",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288253813",
  "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:emergency-disaster-response",
  "citation_count": 4,
  "publication_date": "2026-05-12",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "DORA (Disaster Operational Response Agent benchmark)",
  "artifact_kind": "benchmark",
  "domain": "other:emergency-disaster-response",
  "goal_types": "perform disaster perception from heterogeneous geospatial imagery; conduct spatial relational analysis over roads/population/facilities; plan rescue and evacuation operations; reason about temporal evolution of the disaster; synthesize a multi-modal operational report",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The five task dimensions (perception, spatial analysis, evacuation planning, temporal reasoning, report synthesis) build on each other's outputs within a pipeline, and tool-order/argument-grounding errors compound as trajectories lengthen (agent-to-gold gap widens from 7% to 56% on long pipelines).",
  "n_goals": "515 expert-authored tasks; gold trajectories totaling 3,500 tool-call steps (~6.8/task avg)",
  "tracking_demand": "The agent must compose calls across a 108-tool MCP library over heterogeneous, multi-temporal geospatial data, tracking intermediate perception/analysis outputs and their correctness as the pipeline progresses through the five task dimensions.",
  "scoring": "other:accuracy-against-expert-verified-gold-trajectories",
  "horizon_value": "3,500 tool-call steps total (across 515 tasks, ~6.8/task average)",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "aggregate across the benchmark; implies an average per-task horizon of ~6.8 tool calls, with the paper noting the agent-to-gold gap widens specifically on longer pipelines",
  "horizon_stated": "yes",
  "headline_result": "13 frontier LLMs evaluated; compositional fragility scales with trajectory length, with the agent-to-gold gap widening from 7% to 56% on long pipelines; gold tool-order hints improve accuracy by only 1.08-4.40%.",
  "availability": null,
  "goal_span": "we introduce Disaster Operational Response Agent benchmark (DORA), the first agentic benchmark for end-to-end disaster response: 515 expert-authored tasks across 45 real-world disaster events spanning 10 types, paired with expert-verified, replayable gold trajectories totaling 3,500 tool-call steps. Tasks span five dimensions that cover the operational disaster-response pipeline: disaster perception, spatial relational analysis, rescue and evacuation planning, temporal evolution reasoning, and multi-modal report synthesis.",
  "horizon_span": "515 expert-authored tasks across 45 real-world disaster events spanning 10 types, paired with expert-verified, replayable gold trajectories totaling 3,500 tool-call steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288253813",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290443387",
  "title": "DocOps: A Verifiable Benchmark for Autonomous Agents in Complex Document Operations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-07-22",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "DocOps",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "perform atomic document-operation actions correctly (e.g. edits without destructive metadata corruption); complete escalating, more complex, more tightly coupled document-workflow tasks built from those atomic operations; maintain long-term state tracking and global document consistency across the workflow; correctly verify (not just superficially check) that a document edit was semantically achieved",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The taxonomy explicitly deconstructs document operations into 'atomic dimensions and escalating workflow complexities,' meaning higher-complexity, longer-range tasks are built by composing atomic operations, so an error or destructive edit at an atomic step can break global document consistency required by the larger workflow.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain long-term state tracking of the document's evolving structure/content and verify (rather than superficially assume) that each operation was semantically correct, to preserve global document consistency across highly coupled, long-range tasks.",
  "scoring": "other:deterministic-verifiable-scoring. DocOps is 'a deterministically verifiable evaluation framework,' each operation/task outcome is checked programmatically against ground truth rather than judged qualitatively, though the abstract does not describe an explicit subgoal-checkpoint partial-credit scheme distinct from this verifiability.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we introduce DocOps, a deterministically verifiable evaluation framework underpinned by a hierarchical taxonomy that deconstructs document operations inspired by real-world practices into atomic dimensions and escalating workflow complexities... a fine-grained analysis of existing agents' manipulation behaviors uncovers 3 key failure modes: long-term state tracking collapse, shallow semantic verification, and destructive editing of structural metadata.",
  "horizon_span": "a hierarchical taxonomy that deconstructs document operations inspired by real-world practices into atomic dimensions and escalating workflow complexities",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290443387",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291383324",
  "title": "DuMateBench: Evaluating Autonomous Agents in Complex Real-World Workflows",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-27",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "DuMateBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 200 real-session-derived tasks spanning 8 broad scenarios and 17 fine-grained capability categories; coordinate multiple capability categories within a single task; preserve and correctly use persistent configurations and workspace state carried over from prior interaction history; maintain performance under injected real-world environmental complexity (Insufficient, Unstable, Noisy conditions)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each task preserves relevant pre-solution interaction history, persistent configurations, and workspace state, so completing the task correctly requires reusing/respecting that carried-over context, and most tasks require coordinating multiple of the 17 fine-grained capability categories together rather than exercising just one.",
  "n_goals": "200 tasks spanning 8 broad scenarios and 17 fine-grained capability categories",
  "tracking_demand": "The agent must track persistent configurations, workspace state, and pre-solution interaction history carried into each task, while coordinating multiple capability categories under injected Insufficient, Unstable, or Noisy environmental perturbations.",
  "scoring": "other:hybrid-deterministic-and-LLM-judge-with-robustness-analysis. Performance is assessed via a hybrid deterministic and LLM-as-Judge evaluation protocol, plus complementary robustness, efficiency, and diagnostic analyses under perturbation conditions, beyond one strict binary completion score, though strict task completion itself is also reported and shows substantial gaps.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://dumatebench.com/",
  "goal_span": "The resulting benchmark comprises 200 tasks spanning 8 broad scenarios and 17 fine-grained capability categories, with most tasks requiring multiple capability coordination. We execute these tasks in isolated Docker containers injected with three forms of real-world environmental complexity: Insufficient, Unstable, and Noisy, and assess performance using a hybrid deterministic and LLM-as-Judge evaluation protocol.",
  "horizon_span": "Each task preserves the relevant pre-solution interaction history, persistent configurations, and workspace state, and is then validated through human verification.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291383324",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288671883",
  "title": "Anchor: Mitigating Artifact Drift in Agent Benchmark Generation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-05-25",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "ERP-Bench (via the Anchor task-generation pipeline)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "satisfy explicit procurement-workflow constraints in a production-grade ERP system; satisfy explicit manufacturing-workflow constraints in a production-grade ERP system; reach the solver-certified fully optimal end-state solution for a long-horizon business task, not merely a feasible one",
  "goal_origin": "given-up-front",
  "decomposition": "other:constraint-optimization-program-with-controlled-difficulty-parameters",
  "interdependence": "Each task is generated from a single parametric specification whose constraints jointly define feasibility and optimality, so satisfying some constraints while missing others yields a feasible-but-suboptimal outcome rather than full success; altering parameters changes both difficulty and the interdependence structure.",
  "n_goals": "300 long-horizon tasks (ERP-Bench) spanning procurement and manufacturing workflows",
  "tracking_demand": "The agent must track state-based verifier conditions tied to the end-state of a production-grade ERP system across a long-horizon procurement or manufacturing workflow, since reward depends solely on end-state business correctness rather than intermediate steps.",
  "scoring": "other:constraint-satisfaction-rate-vs-fully-optimal-solution-rate",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier models satisfy explicit task constraints in only 26.1% of ERP-Bench trials and reach a fully optimal solution in just 17.4%, with no human/expert baseline given (comparison is against the solver-certified ground-truth optimum).",
  "availability": "erpbench.ai",
  "goal_span": "we find that generation parameters predict realized difficulty, and that frontier models satisfy explicit task constraints in 26.1% of trials but reach a fully optimal solution in only 17.4% of trials",
  "horizon_span": "AI agents are beginning to complete valuable, long-horizon business operations tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288671883",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291767575",
  "title": "ERPBench: Evaluating LLM Agents for Enterprise Decision-Making Across Competitive Market Ecologies",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-09-04",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "ERPBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "make coupled decisions each round across pricing, production, procurement, inventory, and finance; maximize valuation/rank while competing against fixed rule-based opponents (Solo) or other evaluated LLM agents (Arena) in a shared market; sustain a coherent business strategy across all six rounds of the ERP simulation",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Pricing, production, procurement, inventory, and finance decisions are explicitly 'coupled' within each round, and rounds are sequential within a six-round simulation, so a decision in one area (e.g. overproduction) constrains what is feasible/optimal in another area (e.g. finance) both within and across rounds; in Arena mode, decisions also interact through shared-market competition with other evaluated agents.",
  "n_goals": "100 fixed problems, each run for 6 rounds, in 2 ecologies (Solo, Arena), across 6 model families (7,200 decision rounds total across 1,200 trajectories)",
  "tracking_demand": "Agent must track its own accumulated valuation, inventory, and finance position across all six rounds, plus (in Arena mode) the shared market state shaped by five other competing LLM agents, to decide coupled pricing/production/procurement actions each round.",
  "scoring": "other:mixed \u2014 reports continuous outcome measures (mean valuation, mean rank) per trajectory rather than binary success, with rank-based comparison across ecologies serving as the scoring mechanism; no explicit within-round partial credit beyond the coupled decision outcomes themselves is described.",
  "horizon_value": "6",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per trajectory/problem run)",
  "horizon_stated": "yes",
  "headline_result": "DeepSeek leads in Solo (252.29M mean valuation; mean rank 1.67) while Gemini leads in Arena (263.95M; mean rank 1.76); the two ecologies agree on the task-level winner for only 21 of 100 problems (no human baseline given).",
  "availability": "https://github.com/GAIR-NLP/erp-bench",
  "goal_span": "We introduce ERPBench, an execution-instrumented benchmark for enterprise decision agents in a six-round Enterprise Resource Planning (ERP) simulation with coupled pricing, production, procurement, inventory, finance, and shared-market competition.",
  "horizon_span": "a six-round Enterprise Resource Planning (ERP) simulation with coupled pricing, production, procurement, inventory, finance, and shared-market competition",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291767575",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285462851",
  "title": "EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-02-10",
  "months_since_pub": 7,
  "citations_per_month": 0.14,
  "artifact_name": "EcoGym",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "maintain the Vending sub-environment's profitability (net worth, income) as in Vending-Bench; acquire and retain freelance income under budgeted actions (Freelance sub-environment); sustain operational metrics such as DAU under partial observability (Operation sub-environment); maintain long-term strategic coherence across an effectively unbounded, budgeted-action economic horizon",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "All three environments share a unified decision-making process with standardized interfaces and budgeted actions; because actions are budgeted over a very long (1000+ step) horizon, spending an action early forecloses its use later, and business-relevant outcomes (net worth, income, DAU) accumulate across the whole run rather than resetting per decision.",
  "n_goals": "three sub-environments (Vending, Freelance, Operation)",
  "tracking_demand": "The agent must track budgeted actions, business-relevant outcome metrics (net worth, income, DAU), and partial observability/stochasticity across an effectively unbounded horizon of 1000+ steps (up to 365 day-loops) per evaluation.",
  "scoring": "continuous-reward",
  "horizon_value": "1000+ steps over 365 simulated day-loops (evaluation horizon)",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run",
  "horizon_stated": "yes",
  "headline_result": "Across eleven leading LLMs, no single model dominates across all three scenarios; models show significant suboptimality in either high-level strategy or efficient action execution; no human baseline given.",
  "availability": null,
  "goal_span": "EcoGym comprises three diverse environments: Vending (adapted from the closed-source Vending-Bench, with full open-source release), Freelance (new), and Operation (new), implemented in a unified decision-making process with standardized interfaces, and budgeted actions over an effectively unbounded horizon (1000+ steps if 365 day-loops for evaluation). The evaluation of EcoGym is based on business-relevant outcomes (e.g., net worth, income, and DAU), targeting long-term strategic coherence and robustness under partial observability and stochasticity.",
  "horizon_span": "budgeted actions over an effectively unbounded horizon (1000+ steps if 365 day-loops for evaluation)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285462851",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285050044",
  "title": "EntWorld: A Holistic Environment and Benchmark for Verifiable Enterprise GUI Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 5,
  "publication_date": "2026-01-25",
  "months_since_pub": 8,
  "citations_per_month": 0.62,
  "artifact_name": "EntWorld",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 1,756 tasks spanning six representative enterprise domains (CRM, ITIL, ERP, etc.); operate under strict business logic constraints and high-density enterprise UIs; maintain precise, state-consistent information retrieval verified via SQL-based state-transition checks; complete synthesized long-horizon workflows reverse-engineered from database schemas",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks are schema-grounded workflows whose correctness is checked via SQL-based state-transition validation, so each action must leave the underlying enterprise database in the correct sequential state for the next step/verification to succeed.",
  "n_goals": "1,756 tasks across six enterprise domains",
  "tracking_demand": "The agent must track precise, state-consistent information across a high-density enterprise UI and the underlying database schema, since success is verified deterministically via SQL-based state-transition checks rather than visual matching.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "State-of-the-art GPT-4.1 achieves 47.61% success rate on EntWorld, substantially lower than human performance (exact human score not given in the abstract).",
  "availability": null,
  "goal_span": "Experimental results demonstrate that state-of-the-art models (e.g., GPT-4.1) achieve 47.61% success rate on EntWorld, substantially lower than the human performance, highlighting a pronounced enterprise gap in current agentic capabilities.",
  "horizon_span": "enabling the synthesis of realistic, long-horizon workflows",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285050044",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286775672",
  "title": "Can LLM Agents Be CFOs? Benchmarking Long-Horizon Resource Allocation in an Uncertain Enterprise Environment",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-03-24",
  "months_since_pub": 6,
  "citations_per_month": 0.33,
  "artifact_name": "EnterpriseArena",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "manage liquidity across the firm's operations; close financial books accurately on a regular cycle; gather costly signals about the macro/industry environment before acting; request equity or debt financing appropriately as conditions change; survive (avoid insolvency/failure) across the full 132-month horizon under shifting macroeconomic regimes",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The setting requires 'binding commitments under partial observability, delayed consequences, hard resource budgets, and shifting dynamics,' so financing/signal-gathering decisions made in earlier months have delayed consequences that constrain liquidity and solvency in later months, and failures cascade across observation, action timing, and capital sizing.",
  "n_goals": "132-month simulated horizon; 23 LLMs and 4 agent frameworks evaluated",
  "tracking_demand": "The agent must track liquidity, capital structure (equity/debt), and signals about shifting macroeconomic/industry regimes month by month across the 132-month horizon, since consequences of earlier decisions are delayed and only become apparent later.",
  "scoring": "other:survival-rate-with-cascading-failure-analysis. Only 15.4% of trials survive the full 132-month horizon, and the paper attributes failures to cascades 'across observation, action timing, and capital sizing' -- process-level failure attribution alongside the binary survival outcome, though not a formal subgoal-checkpoint rubric.",
  "horizon_value": "132-month CFO simulation",
  "horizon_unit": "other:simulated-months",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole 132-month simulated run)",
  "horizon_stated": "yes",
  "headline_result": "Only 15.4% of trials (across 23 LLMs and 4 agent frameworks) survive the full 132-month horizon, and larger models do not reliably outperform smaller ones; no human/expert baseline given.",
  "availability": null,
  "goal_span": "We introduce EnterpriseArena, a 132-month CFO simulator that evaluates long-horizon resource allocation under uncertainty in a FinTech lending firm. Agents must manage liquidity, close books, gather costly signals, and request equity or debt financing across changing macroeconomic regimes.",
  "horizon_span": "We introduce EnterpriseArena, a 132-month CFO simulator that evaluates long-horizon resource allocation under uncertainty in a FinTech lending firm.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286775672",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286766304",
  "title": "EnterpriseLab: A Full-Stack Platform for developing and deploying agents in Enterprises",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-03-23",
  "months_since_pub": 6,
  "citations_per_month": 0.17,
  "artifact_name": "EnterpriseArena (via the EnterpriseLab platform)",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "complete complex enterprise workflows spanning IT, HR, sales, and engineering domains; correctly invoke and sequence tool calls across 140+ tools exposed via Model Context Protocol across 15 applications; match frontier-model performance while running as a smaller (8B) privacy-preserving model; generalize/remain robust across diverse enterprise benchmarks (EnterpriseBench, CRMArena)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Enterprise workflows require orchestrating many interdependent tools across 15 applications and 4 domains (IT, HR, sales, engineering) exposed via a shared Model Context Protocol layer, so a workflow's tool calls must be sequenced correctly with respect to shared enterprise data/state.",
  "n_goals": "15 applications; 140+ tools across IT, HR, sales, and engineering domains",
  "tracking_demand": "The trained agent must track which of many interdependent enterprise tools/applications it has invoked and their resulting state, to complete complex multi-tool enterprise workflows without needing frontier-scale model capacity.",
  "scoring": "other:performance-parity-with-GPT-4o-plus-cost-reduction-and-cross-benchmark-generalization",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "8B-parameter models trained within EnterpriseLab match GPT-4o's performance on complex enterprise workflows while reducing inference costs by 8-10x, and improve +10% on EnterpriseBench and CRMArena; no explicit human/expert baseline is given (comparison is model-vs-model).",
  "availability": null,
  "goal_span": "Our results demonstrate that 8B-parameter models trained within EnterpriseLab match GPT-4o's performance on complex enterprise workflows while reducing inference costs by 8-10x, and remain robust across diverse enterprise benchmarks, including EnterpriseBench (+10%) and CRMArena (+10%).",
  "horizon_span": "We validate the platform through EnterpriseArena, an instantiation with 15 applications and 140+ tools across IT, HR, sales, and engineering domains.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286766304",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289621469",
  "title": "EnterpriseClawBench: Benchmarking Agents from Real Workplace Sessions",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-06-22",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "EnterpriseClawBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "read heterogeneous workplace files relevant to a task; invoke the correct tools to accomplish a business objective; deliver a business artifact matching role-specific hard rules and semantic rubrics",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Delivering the required business artifact depends on first correctly reading the right heterogeneous files and invoking tools consistent with the task's hard rules; a semantic rubric then checks that the final artifact reflects those earlier steps correctly.",
  "n_goals": "852 reproducible tasks, each with recovered fixtures, role classes, skill subclasses, hard rules, and semantic rubrics",
  "tracking_demand": "The agent must track which files it has read, what tools it has invoked, and whether its eventual delivered artifact satisfies the task's hard rules and semantic rubric, since harness-model combination, artifact delivery, and visual quality are all separately reported.",
  "scoring": "other:rubric-and-hard-rule-compliance. The paper reports a single best-configuration score (0.663) and argues evaluation 'must report harness--model combinations, artifact delivery, visual quality, cost, runtime, and skill-transfer behavior, rather than collapsing performance into a single score,' implying multi-dimensional but not an explicit subgoal-checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best configuration (Codex with GPT-5.5) reaches only 0.663 on EnterpriseClawBench; no human/expert baseline reported in the abstract.",
  "availability": "https://github.com/FrontisAI/EnterpriseClawBench",
  "goal_span": "Starting from a large archive of workplace sessions, the EnterpriseClawBench produces 852 reproducible tasks, each paired with recovered fixtures, rewritten prompts, role classes, skill subclasses, hard rules, and semantic rubrics.",
  "horizon_span": "the EnterpriseClawBench produces 852 reproducible tasks, each paired with recovered fixtures, rewritten prompts, role classes, skill subclasses, hard rules, and semantic rubrics.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289621469",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286568645",
  "title": "EnterpriseOps-Gym: Environments and Evaluations for Stateful Agentic Planning and Tool Use in Enterprise Settings",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 11,
  "publication_date": "2026-03-13",
  "months_since_pub": 6,
  "citations_per_month": 1.83,
  "artifact_name": "EnterpriseOps-Gym",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 1,150 expert-curated enterprise tasks across eight mission-critical verticals (e.g., Customer Service, HR, IT); plan and act correctly amid persistent state changes and strict access-control protocols; correctly refuse infeasible tasks rather than attempting unintended, potentially harmful side effects",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Actions operate on 164 shared database tables via 512 tools with strict access protocols, so an earlier action's persistent state change (or an access violation) constrains what later actions in the same task are valid or safe to take.",
  "n_goals": "1,150 expert-curated tasks across eight mission-critical verticals",
  "tracking_demand": "Agent must track persistent state changes across 164 database tables, access-control constraints, and whether the current task is feasible at all, to avoid unintended side effects from attempting an infeasible task.",
  "scoring": "other:success-rate-plus-refusal-rate - top model (Claude Opus 4.5) achieves only 37.4% task success; oracle human plans improve performance by 14-35 points; agents are also scored on correctly refusing infeasible tasks (best model 53.9%); a multi-dimensional scoring scheme rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "The top-performing Claude Opus 4.5 achieves only 37.4% success across 1,150 tasks; oracle human plans improve performance by 14-35 percentage points, pinpointing strategic reasoning as the primary bottleneck; agents frequently fail to refuse infeasible tasks (best model achieves only 53.9%).",
  "availability": null,
  "goal_span": "EnterpriseOps-Gym features a containerized sandbox with 164 database tables and 512 functional tools to mimic real-world search friction. Within this environment, agents are evaluated on 1,150 expert-curated tasks across eight mission-critical verticals (including Customer Service, HR, and IT).",
  "horizon_span": "EnterpriseOps-Gym features a containerized sandbox with 164 database tables and 512 functional tools to mimic real-world search friction.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286568645",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288653469",
  "title": "Evaluating Deep Research Agents on Expert Consulting Work: A Benchmark with Verifiers, Rubrics, and Cognitive Traps",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-05-17",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": null,
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "produce a multi-document, decision-grade consulting deliverable from SME-authored prompts; pass deterministic binary verifiers (mean 14.9 per task); satisfy each of a five-criterion SME rubric (Data Integrity, Analytical Rigor, Relevance & Focus, Execution Precision, Format & Deliverability); avoid embedded cognitive traps (human-error mimicry, deterministic precision traps) that penalize surface-pattern reasoning",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Verifiers and rubric criteria are scored jointly per task via a combined threshold (rubric mean >= 2.5 and verifier pass rate >= 80%), so an agent can satisfy many verifiers yet still fail acceptance if it misses a rubric dimension or falls for an embedded trap.",
  "n_goals": "mean 14.9 binary verifiers per task, plus a 5-criterion rubric, across 70 prompts",
  "tracking_demand": "The agent must track which of the ~14.9 verifiers per task it has satisfied, avoid embedded cognitive traps requiring reconciliation against context, and keep its final deliverable consistent with all five rubric criteria simultaneously.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Acceptance rates uniformly low: o3 15.7%, Claude 12.9%, Gemini 12.9%; on continuous VRS, o3 leads at 61.4 (vs Gemini 52.6, Claude 38.5); no human-baseline VRS given.",
  "availability": "code, evaluation code, and full prompt corpus publicly released (no URL given in abstract)",
  "goal_span": "we score two complementary layers: deterministic binary verifiers (mean 14.9 per task) and a five-criterion 0--3 SME rubric ... combined into a Verifier-Rubric Score (VRS, 0--100). Acceptance under a joint threshold (rubric mean >= 2.5 and verifier pass rate >= 80%) is uniformly low",
  "horizon_span": "a benchmark of 70 SME-authored management consulting prompts, each embedding cognitive traps that penalize surface-pattern reasoning",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288653469",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284704177",
  "title": "The Agent's First Day: Benchmarking Learning, Exploration, and Scheduling in the Workplace Scenarios",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-01-13",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "EvoEnv",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "schedule and prioritize streaming tasks with varying priorities under a dynamic workload; actively seek out information to reduce hallucination before acting (prudent information acquisition); distill and reuse generalized strategies learned from earlier dynamically generated tasks in later ones (continuous evolution)",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "Later performance depends on strategies distilled from earlier dynamically generated tasks (continuous evolution), and streaming tasks compete for the agent's attention/scheduling under varying priorities, so mishandling earlier tasks or mis-prioritizing can degrade later scheduling and learning.",
  "n_goals": "50 dynamic scenarios, each with 2-6 task instances (meta-tasks)",
  "tracking_demand": "The agent must track incoming streaming tasks and their priorities, its own accumulated exploration/experience for continual learning, and its confidence in acquired information to avoid hallucination, across a continuously evolving workplace scenario.",
  "scoring": "other:three-axis-capability-scoring. The abstract evaluates agents 'along three dimensions' (context-aware scheduling, prudent information acquisition, continuous evolution) rather than a single milestone/subgoal-checkpoint rubric.",
  "horizon_value": "50 dynamic scenarios, each with 2-6 task instances; observed usage up to ~90 steps and 232 tool calls for the highest-usage evaluated model (Gemini-3-Flash)",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per scenario run; step/tool-call counts vary substantially by evaluated model",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/KnowledgeXLab/EvoEnv",
  "goal_span": "\\method{} evaluates agents along three dimensions: (1) context-aware scheduling for streaming tasks with varying priorities; (2) prudent information acquisition to reduce hallucination via active exploration; and (3) continuous evolution by distilling generalized strategies from rule-based, dynamically generated tasks.",
  "horizon_span": "Each scenario encompasses 2 to 6 task instances... While Gemini-3-Flash uses substantially more steps (90) and tool calls (232) than the middle-tier models, this increased activity reflects the necessary complexity",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284704177",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291217421",
  "title": "FM-Bench: A Benchmark for Long-Horizon Management with Competing Agents",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-19",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "FM-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "draft and manage a football squad within a fixed budget shared with rivals; trade players and negotiate contracts across the season; invest in facilities and youth development; set match lineups; maintain board confidence (avoid being fired) while maximizing a final accumulated score over 20 in-game years",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Investment, trading, contract, and lineup decisions accumulate over 20 in-game years into one final score via a deterministic engine, so early decisions (e.g. slow-payoff investments, cash allocation, contract renewal timing) directly constrain what remains possible in later years; in the Arena track the same shared 20-year world is contested head-to-head against other models plus a scripted anchor.",
  "n_goals": "20 in-game years; 26 tools; roughly 340-400 decision stops per run; 6 measured behavioral capabilities; 15 frontier models (solo) plus an Arena with the same 15 models and a scripted anchor",
  "tracking_demand": "The agent must track squad composition, budget/cash flow, ongoing contract-renewal deadlines, facility/youth investments, and the board's confidence in it, across roughly 340-400 decision stops spanning 20 in-game years, since 'the order settles only late in the horizon.'",
  "scoring": "continuous-reward. 'A deterministic engine accumulates every year into one final score with no LLM judge or human rater,' and the paper separately measures 'six behavioral capabilities behind the score' -- process-level behavioral analysis in addition to the single accumulated final score, though not framed as an explicit milestone/subgoal-checkpoint rubric.",
  "horizon_value": "20 in-game years; roughly 340 to 400 decision stops per run",
  "horizon_unit": "other:decision-stops",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole 20-year managerial run)",
  "horizon_stated": "yes",
  "headline_result": "claude-fable-5 tops both the solo leaderboard (mean score) and the Arena, though the Arena's title 'nonetheless rotates among ten models'; the best first-play human lands only at the bottom of the model board. No single numeric score/gap given in the abstract beyond this qualitative ranking.",
  "availability": "https://github.com/Analogy-AI/fm-bench",
  "goal_span": "An LLM agent runs a football club for 20 in-game years through 26 tools and roughly 340 to 400 decision stops. It drafts a squad on the same budget as every rival, trades players, negotiates contracts, invests in facilities and youth, sets lineups, and answers to a board that can fire it, while a deterministic engine accumulates every year into one final score with no LLM judge or human rater.",
  "horizon_span": "An LLM agent runs a football club for 20 in-game years through 26 tools and roughly 340 to 400 decision stops.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291217421",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291531031",
  "title": "FORESIGHT-9: Prospective and Process-Aware Evaluation of Adaptive Trading Agents",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:financial-trading",
  "citation_count": 0,
  "publication_date": "2026-08-29",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "FORESIGHT-9",
  "artifact_kind": "environment/simulator",
  "domain": "other:financial-trading",
  "goal_types": "maintain a coherent, active factor-ensemble/trading strategy across staged macro-financial events without silent internal collapse; respond adaptively to joint multi-asset anchors and disclosed in-world-time observations across a multi-year counterfactual worldline; keep portfolio holdings and decision-records mutually consistent (avoid decision-record vs. execution divergence)",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each worldline's staged macro-financial events and joint multi-asset anchors constrain later decisions, and decision-record state must stay coherent with actual executed holdings across the run.",
  "n_goals": "nine worldlines x two agent frameworks x two model backbones = 36 long-horizon runs, each spanning roughly 2,468 simulated trading days",
  "tracking_demand": "The agent must track its adaptive factor library/ensemble state, executed portfolio holdings, and staged event disclosures across a multi-year, ~2,468-trading-day counterfactual worldline, ensuring internal decision-records remain consistent with actual executed holdings.",
  "scoring": "other:process-and-outcome-telemetry -- both portfolio outcomes and process telemetry (whether the factor library and executed holdings remain coherent) are scored; the abstract contrasts terminal returns with process-level coherence checks, indicating credit beyond a single final-outcome number, though no explicit rubric/checkpoint scheme is named.",
  "horizon_value": "each worldline runs from the 2026-07-15 information boundary through 2035-12-31, approximately 2,468 simulated trading days (~9 years 5 months) per run, across 36 long-horizon runs total",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/run (one full worldline trajectory)",
  "horizon_stated": "yes",
  "headline_result": "A fixed equal-weight policy outperforms 31 of 36 agent runs; no specific best-agent-vs-human/expert performance gap is reported (comparison is agent-vs-simple-baseline, not vs. human traders).",
  "availability": null,
  "goal_span": "We evaluate two adaptive trading-agent frameworks with two foundation-model backbones across 36 long-horizon runs. Agent rankings vary substantially across worldlines and backbones, and a fixed equal-weight policy outperforms 31 of 36 runs. Process telemetry exposes failures that terminal returns conceal: in one high-return run, the live factor library collapsed while executed holdings converged to the equal-weight fallback, even though decision records continued to report an active factor ensemble.",
  "horizon_span": "each run terminates at 2035-12-31 ... approximately 2,468 daily marks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291531031",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290856801",
  "title": "FinEvo-Bench: A Longitudinal Benchmark for Self-Evolving Agents in Professional Financial Workflows",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-08-06",
  "months_since_pub": 1,
  "citations_per_month": 1.0,
  "artifact_name": "FinEvo-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 120 real-case-grounded financial tasks across 20 business scenes in six financial domains, following institution-provided procedures and constraints; retain and re-apply experience from earlier cases in a scene to later, substantively distinct cases sharing the same professional procedure (self-evolution); maintain financial-compliance quality across each task as scored by a manually reviewed rubric; sustain or improve quality/compliance across an interleaved, shuffled task stream over the longitudinal run",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "other:six-related-cases-per-scene-with-longitudinal-cross-task-experience-transfer",
  "interdependence": "Cases within a scene share a professional procedure, so experience distilled from earlier cases in the same scene or globally interleaved stream can and should transfer to later cases; paired non-evolving controls show this transfer is real (evolving conditions gain 9.33-19.37 points over non-evolving), meaning tasks are not fully independent despite being run as separate cases.",
  "n_goals": "120 tasks; 20 business scenes across six financial domains; each scene has six related but distinct cases; three independently shuffled, globally interleaved task streams",
  "tracking_demand": "The agent (self-evolving scaffold) must retain experience -- via memory, skill distillation, or both -- from earlier cases in the interleaved task stream and apply it to later, procedurally-related but factually distinct cases, while satisfying institution-provided professional-procedure constraints and financial-compliance requirements on each individual task.",
  "scoring": "milestone-rubric",
  "horizon_value": "120 real-case-grounded tasks per longitudinal stream (20 scenes x 6 cases each); three independently shuffled, globally interleaved task streams",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run (the longitudinal task stream over which self-evolution is measured)",
  "horizon_stated": "yes",
  "headline_result": "Letta achieves the highest evolved score (91.65) and fewest compliance issues (0.09/task); Codex achieves the largest self-evolution gain (+19.37); across scaffolds, evolving conditions raise scores by 9.33-19.37 points and reduce compliance issues by 0.12-0.44 per task versus non-evolving controls. No human/expert professional baseline score is given.",
  "availability": null,
  "goal_span": "Each scene contains six related but substantively distinct cases that share a professional procedure and a manually reviewed rubric for task quality and financial compliance... Letta achieves the highest evolved score (91.65) and fewest compliance issues (0.09 per task); Codex achieves the largest self-evolution gain (+19.37).",
  "horizon_span": "We introduce FinEvo-Bench, a longitudinal benchmark with 120 real-case-grounded tasks, 20 business scenes across six financial domains... We compare four self-evolving agent scaffolds using the same Qwen3.7-Max backbone and three independently shuffled, globally interleaved task streams.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290856801",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287209412",
  "title": "FrontierFinance: A Long-Horizon Computer-Use Benchmark of Real-World Financial Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-04-07",
  "months_since_pub": 5,
  "citations_per_month": 0.4,
  "artifact_name": "FrontierFinance",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 25 complex financial modeling tasks across five core finance models; produce client-ready outputs matching industry-standard financial-modeling workflows; satisfy detailed structured evaluation rubrics per task",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Complex financial-modeling tasks require multi-step reasoning across five core finance models; because tasks reflect industry-standard workflows requiring an average of over 18 hours of skilled human labor, later modeling steps depend on the correctness of earlier assumptions/build-out.",
  "n_goals": "25 complex financial modeling tasks across five core finance models",
  "tracking_demand": "The agent must track detailed financial-modeling state (assumptions, formulas, model linkages) across a task requiring on average over 18 hours of equivalent skilled human labor, checked against detailed structured rubrics.",
  "scoring": "other:rubric-graded-with-human-expert-comparison. Tasks are paired with detailed rubrics for structured evaluation, and human experts both perform the tasks as baselines and grade LLM outputs, giving explicit rubric-based (non-binary) grading plus a direct human-baseline comparison.",
  "horizon_value": "average of over 18 hours of skilled human labor per task",
  "horizon_unit": "human-expert-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Human experts both receive higher average scores and are more likely to produce client-ready outputs than current state-of-the-art systems; no single top-model number is given in the abstract.",
  "availability": null,
  "goal_span": "we introduce FrontierFinance, a long-horizon benchmark of 25 complex financial modeling tasks across five core finance models, requiring an average of over 18 hours of skilled human labor per task to complete. Developed with financial professionals, the benchmark reflects industry-standard financial modeling workflows and is paired with detailed rubrics for structured evaluation.",
  "horizon_span": "requiring an average of over 18 hours of skilled human labor per task to complete",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287209412",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "human-expert-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290836629",
  "title": "GDPevo: Evaluating Agent Self-Evolution on Real Business Tasks",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-04",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "GDPevo",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "update an agent's persistent state from prior training-task experience (self-evolution); apply recombined atomic business rules correctly on held-out test tasks across CRM/ERP/finance/healthcare/legal/data-centric workflows; achieve held-out accuracy gains attributable specifically to training experience rather than data contamination",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Atomic business rules are distributed across training tasks and recombined in held-out test tasks, so test-time success on a rule combination requires that the relevant subset of rules was individually learned from disjoint training tasks; getting one atomic rule wrong can break any task recombining it.",
  "n_goals": "V1: 120 tasks in 12 groups (5 training + 5 held-out test tasks per group); V2: 240 tasks in 24 groups",
  "tracking_demand": "The evaluation harness must track which atomic business rules were exposed during training versus recombined at test time, per task group, to attribute held-out gains specifically to training experience rather than contamination.",
  "scoring": "other:held-out-accuracy-gain-attributable-to-training. Self-evolution is measured by held-out accuracy improvement (up to +16.44 percentage points) against a fully-informed oracle ceiling (91.6%), an explicit before/after, rule-attributable partial-credit design rather than a single static pass/fail score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Self-evolution improves held-out accuracy by up to 16.44 percentage points, but the best evolved agents remain far below the fully informed oracle ceiling of 91.6%; no human/expert baseline given.",
  "availability": "https://github.com/Prism-Shadow/GDPevo",
  "goal_span": "Its core mechanism, rule hybridization, decomposes each enterprise workflow into atomic business rules, distributes subsets of these rules across training tasks, and recombines them in held-out test tasks so that test-time gains are attributable.",
  "horizon_span": "V1 release contains 120 tasks in 12 groups, with five training and five held-out test tasks per group",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290836629",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285303555",
  "title": "H-AdminSim: A Multi-Agent Simulator for Realistic Hospital Administrative Workflows with FHIR Integration",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-02-05",
  "months_since_pub": 7,
  "citations_per_month": 0.0,
  "artifact_name": "H-AdminSim",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "process hospital administrative requests (drawn from a workload of 10,000+ requests/day in large hospitals); coordinate across multiple administrative subtasks rather than handling them in isolation; operate correctly against FHIR-integrated, heterogeneous hospital-setting data",
  "goal_origin": "given-up-front",
  "decomposition": "other:multi-agent-simulated-administrative-workflows-not-fully-specified-in-abstract",
  "interdependence": "The abstract frames prior work as failing because it treats administrative subtasks in isolation, implying H-AdminSim's workflows instead require subtasks to interact/coordinate, though the exact coupling mechanism is not detailed in the abstract.",
  "n_goals": null,
  "tracking_demand": "The multi-agent simulation must track hospital administrative request state across heterogeneous hospital settings via a unified FHIR-integrated environment, evaluated against detailed rubrics.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "These tasks are quantitatively evaluated using detailed rubrics, enabling systematic comparison of LLMs.",
  "horizon_span": "in large hospitals, process over 10,000 requests per day",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285303555",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290634770",
  "title": "HANDBOOK.md: A Benchmark for Long-Context Agentic Instruction Following",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-07-28",
  "months_since_pub": 2,
  "citations_per_month": 1.0,
  "artifact_name": "HANDBOOK.md",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "locate the specific clauses in a long standing-policy document (20-124 pages) that apply to the current situation; carry out routine professional work strictly governed by that policy document across many tool-mediated actions; avoid letting a plausible but unauthorized in-environment request override the standing policy; satisfy every one of many deterministic, programmatic grading criteria (824 total) simultaneously",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Completing a task correctly requires holding many interacting policy clauses in mind simultaneously across a long tool-use horizon, so satisfying one rubric criterion can be undermined by an in-environment request that appears to override an earlier standing policy, and by failing to perform a policy-mandated check at the right point.",
  "n_goals": "65 agentic tasks; a rubric of 824 total programmatic criteria (required and prohibited actions), each task governed by a 20-124 page SOP",
  "tracking_demand": "The agent must hold the relevant clauses of a long standing-policy document across roughly 17 reasoning steps and 30 tool calls on average, checking each action against required and prohibited criteria rather than losing rule details over the long horizon.",
  "scoring": "milestone-rubric. 'Grading is fully deterministic: each task carries a rubric of programmatic criteria (824 in total) that check both that required actions occurred and that prohibited actions did not,' and 'a trial passes only if every criterion is satisfied' -- explicit per-criterion (subgoal-level) checking underlying a strict overall pass/fail.",
  "horizon_value": "~17 reasoning steps and ~30 tool calls on average per task",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "The strongest evaluated model passes 36.2% of trials under strict grading (every criterion satisfied), and most frontier models remain below 25%; no human/expert baseline given.",
  "availability": null,
  "goal_span": "We present HANDBOOK_md, a benchmark of 65 agentic tasks modeled on how employees follow company handbooks. Each task places an agent in a self-contained company environment... and instructs it to carry out routine professional work governed by an expert-written standard operating procedure of 20-124 pages... each task carries a rubric of programmatic criteria (824 in total) that check both that required actions occurred and that prohibited actions did not.",
  "horizon_span": "Completing a task requires locating the clauses that apply, holding them across a horizon of roughly 17 reasoning steps and 30 tool calls on average.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290634770",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287425604",
  "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 8,
  "publication_date": "2026-04-10",
  "months_since_pub": 5,
  "citations_per_month": 1.6,
  "artifact_name": "HealthAdminBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete a Prior Authorization workflow end-to-end across an EHR and payer portal; complete an Appeals and Denials Management workflow; complete a Durable Medical Equipment (DME) Order Processing workflow; satisfy each of the fine-grained verifiable subtasks a task decomposes into (often 15+ per task)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each of the 135 tasks decomposes into fine-grained subtasks across four connected GUI environments (EHR, two payer portals, fax system); many tasks involve 15 or more subtasks whose order and cross-app data consistency must be preserved for the workflow to reach a valid end state.",
  "n_goals": "135 tasks yielding 1,698 evaluation points (many tasks involve 15+ subtasks)",
  "tracking_demand": "The agent must track progress across many fine-grained, cross-application subtasks (spanning EHR, payer portals, and fax) per administrative workflow, under a fixed interaction budget, to reach a correct terminal workflow state.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Best-performing agent (Claude Opus 4.6 CUA) achieves only 36.3% task success, while GPT-5.4 CUA attains the highest subtask success rate (82.8%); no human baseline given.",
  "availability": "https://github.com/som-shahlab/health-admin-bench",
  "goal_span": "Each task is decomposed into fine-grained, verifiable subtasks, yielding 1,698 evaluation points. ... the best-performing agent (Claude Opus 4.6 CUA) achieves only 36.3 percent task success, while GPT-5.4 CUA attains the highest subtask success rate (82.8 percent).",
  "horizon_span": "Since many tasks involve 15 or more subtasks ... agents operate 'under a fixed interaction budget' and references 'maximum steps,' but does not specify an exact number.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287425604",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288253651",
  "title": "Herculean: An Agentic Benchmark for Financial Intelligence",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 3,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "Herculean",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete Trading workflow tasks via a standardized MCP-based skill environment; complete Hedging workflow tasks requiring long-horizon coordination and state consistency; complete Market Insights workflow tasks; complete Auditing workflow tasks requiring structured verification",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each of the four workflows (Trading, Hedging, Market Insights, Auditing) is its own MCP-based skill environment with distinct tools and success criteria, but within Hedging and Auditing specifically, long-horizon coordination and state consistency across multiple steps are called out as critical, meaning later steps depend on correctly maintained state/verification from earlier steps.",
  "n_goals": "4 representative financial workflows (Trading, Hedging, Market Insights, Auditing)",
  "tracking_demand": "The agent must track workflow-specific state, tool interactions, and constraints within each MCP-based skill environment, with Hedging and Auditing additionally requiring sustained state consistency and structured verification across long-horizon coordination.",
  "scoring": "other:per-workflow-success-criteria. Each workflow is instantiated with its own standardized success criteria enabling consistent end-to-end assessment -- a per-workflow (subgoal-domain) breakdown rather than one single aggregate score; agents perform well on Trading/Market Insights but poorly on Hedging/Auditing.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce Herculean, the first skilled benchmark for agentic financial intelligence spanning four representative workflows, including Trading, Hedging, Market Insights, and Auditing. Each workflow is instantiated as a standardized MCP-based skill environment with its own tools, interaction dynamics, constraints, and success criteria, enabling consistent end-to-end assessment of heterogeneous agent systems.",
  "horizon_span": "where long-horizon coordination, state consistency, and structured verification are critical",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288253651",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288671779",
  "title": "JobBench: Aligning Agent Work With Human Will",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 4,
  "publication_date": "2026-05-25",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "JobBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete 130 agentic tasks spanning 35 occupations that experts identify as high-priority for delegation; reason through cluttered, heterogeneous reference-file information streams within a professional workspace; satisfy a fact-anchored chain of rubrics averaging 35.6 binary criteria per task",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each task's rubric chain requires satisfying an average of 35.6 binary sub-criteria that are fact-anchored to a workspace of heterogeneous reference files, so misreading one reference file can cascade into multiple failed criteria.",
  "n_goals": "130 tasks across 35 occupations; averaging 35.6 binary rubric criteria per task",
  "tracking_demand": "Agent must correctly reason through a workspace of heterogeneous reference files and track satisfaction of an average of 35.6 fact-anchored binary rubric criteria per task.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The strongest of 36 evaluated models, Claude Opus 4.7 under Claude Code, reaches only 45.9% on the fact-anchored rubric criteria; no human baseline given.",
  "availability": null,
  "goal_span": "Each task is packaged as a workspace of heterogeneous reference files, requiring the agent to reason through the cluttered information streams of real professional work. Outputs are graded by a fact-anchored chain of rubrics, averaging 35.6 binary criteria per task.",
  "horizon_span": "JobBench covers 130 agentic tasks across 35 occupations.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288671779",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286774040",
  "title": "LH-Bench: Skill-Grounded Evaluation of Long-Horizon Agents on Subjective Enterprise Tasks",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-03-24",
  "months_since_pub": 6,
  "citations_per_month": 0.33,
  "artifact_name": "LH-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "produce subjective, context-dependent enterprise work whose quality depends on organizational goals and user intent, not a single correct answer; produce correct intermediate artifacts across long, multi-tool workflows (e.g., chapter-level content, Figma-to-code conversions); satisfy expert-grounded rubrics scoring subjective work quality; align with pairwise human preference judgments as convergent validation",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Long multi-tool workflows produce intermediate artifacts (e.g., individual chapters of a course, or individual Figma-to-code conversions) whose quality is scored via stepwise reward signals, so overall task quality depends on the quality of each intermediate artifact along the workflow, not just a final output.",
  "n_goals": "two environments: Figma-to-code (33 real .fig tasks) and Programmatic content (41 courses comprising 183 individually-evaluated chapters)",
  "tracking_demand": "The agent must track and produce correct intermediate artifacts (e.g., each chapter of a course, or each Figma-to-code conversion) across long, multi-tool workflows, since ground-truth artifacts enable stepwise reward signals at that granularity rather than only a single final judgment.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Domain-authored rubrics provide substantially more reliable evaluation signals than LLM-authored rubrics (kappa=0.60 vs. 0.46), with human preference judgments confirming the same top-tier model separation (p<0.05); no single specific top-model accuracy score or human/expert task-performance baseline is given as a headline task-completion number.",
  "availability": null,
  "goal_span": "The pillars are: (i) expert-grounded rubrics that give LLM judges the domain context needed to score subjective work, (ii) curated ground-truth artifacts that enable stepwise reward signals (e.g., chapter-level annotation for content tasks), and (iii) pairwise human preference evaluation for convergent validation.",
  "horizon_span": "Programmatic content (41 courses comprising 183 individually-evaluated chapters on a course platform serving 30+ daily users)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286774040",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290830424",
  "title": "Post-Training on Office Work Improves Software Engineering: A Behavioral Account of Cross-Domain Transfer",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-03",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Long-Horizon Multi-Tool Agent (LHMTA) task collection",
  "artifact_kind": "dataset",
  "domain": "business-office-enterprise",
  "goal_types": "select goals appropriately within nested and branching office work; construct task-relevant state as work unfolds; maintain fidelity to higher-level objectives across nested/branching sub-work; verify completion against the environment",
  "goal_origin": "mixed:given-up-front-tasks-with-self-directed-goal-selection",
  "decomposition": "hierarchical",
  "interdependence": "Nested and branching tasks require goal selection at multiple levels to remain consistent with higher-level objectives, and environment-verification of completion closes the loop determining whether execution can proceed or must be revised.",
  "n_goals": "363 Long-Horizon Multi-Tool Agent (LHMTA) tasks drawn from office workflows",
  "tracking_demand": "Agent must repeatedly select goals, construct task-relevant state, maintain fidelity to higher-level objectives, and verify completion against the environment across nested and branching long-horizon office work.",
  "scoring": "other:matched-trajectory-behavioral-analysis",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Post-training on 363 office-workflow LHMTA tasks (containing no software-engineering tasks) improved pass@1 by 5.8 points on SWE-Bench Pro, with matched trajectory analysis showing gains in all four GDE behaviors; no human baseline given.",
  "availability": null,
  "goal_span": "We call this capability goal-directed execution (GDE): the repeated application of four behaviors, namely selecting goals, constructing task-relevant state, maintaining fidelity to higher-level objectives, and verifying completion against the environment... We test this by post-training Qwen3.5-122B-A10B on 363 Long-Horizon Multi-Tool Agent (LHMTA) tasks drawn from office workflows.",
  "horizon_span": "We test this by post-training Qwen3.5-122B-A10B on 363 Long-Horizon Multi-Tool Agent (LHMTA) tasks drawn from office workflows.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290830424",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287636872",
  "title": "Four-Axis Decision Alignment for Long-Horizon Enterprise AI Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-04-21",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "LongHorizon-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "make high-stakes regulated decisions (loan qualification, insurance claims adjudication) under lossy memory and multi-step reasoning; maintain factual precision (FRP) about case facts over the long horizon; maintain reasoning coherence (RCS) across the multi-step decision process; reconstruct compliance/regulatory justification (CRR) for the decision; calibrate abstention (CAR) -- know when to decline to decide rather than commit incorrectly",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "The four axes (FRP, RCS, CRR, CAR) are explicitly designed to be orthogonal and independently measurable/failable, though the paper finds architecture-level trade-offs across axes (e.g., retrieval collapses on FRP; schema-anchored architectures pay a 'scaffolding tax'), so real systems' performance on one axis is coupled with design choices affecting others.",
  "n_goals": "four alignment axes (FRP, RCS, CRR, CAR); two decisioning domains (loan qualification, insurance claims adjudication); six memory architectures evaluated",
  "tracking_demand": "The agent must retain case facts accurately over a long-horizon multi-step review under lossy memory, maintain a coherent reasoning chain, reconstruct which regulatory rules justify the decision, and calibrate when to abstain rather than commit -- all against deterministic ground truth.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Retrieval-based memory collapses specifically on factual precision (FRP); plain summarization under a fact-preservation prompt is a strong baseline on FRP/RCS/EDA/CRR; all six architectures commit (fail calibrated abstention) on every case; no single top score or human baseline is given.",
  "availability": null,
  "goal_span": "We propose that long-horizon decision behavior decomposes into four orthogonal alignment axes, each independently measurable and failable: factual precision (FRP), reasoning coherence (RCS), compliance reconstruction (CRR), and calibrated abstention (CAR).",
  "horizon_span": "Long-horizon enterprise agents make high-stakes decisions (loan underwriting, claims adjudication, clinical review, prior authorization) under lossy memory, multi-step reasoning, and binding regulatory constraints.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287636872",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290118634",
  "title": "LongMedBench: Benchmarking Medical Agents for Long-Horizon Clinical Decision-Making",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-10",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "LongMedBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "aggregate evidence across a patient's repeated visits, tests, and evolving treatments to answer fact-based QA; perform temporal reasoning over the patient's event stream, including implicit (not just explicit-timestamped) time inference; make a long-horizon clinical decision that correctly uses historical patient information accumulated over many visits",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Because 'real-world medical care is inherently longitudinal', later visits' correct clinical decisions depend on evidence aggregated across all earlier visits/tests/treatments in the same patient's time-series event stream, and the paper finds decision-making performance is 'highly dependent on the model's immediate context', implying degraded performance when earlier history is not properly integrated.",
  "n_goals": "335 patients, avg 19.72 inpatient visits/patient, avg 44.91 medical events/visit; 3 evaluation suites (fact-based QA, temporal reasoning, long-horizon decision-making)",
  "tracking_demand": "Agent must integrate time-series clinical events (admission records, notes) across many visits per patient, correctly distinguish explicit timestamps from cases requiring implicit time inference, and carry this forward into long-horizon decision-making.",
  "scoring": "other:not-stated precisely \u2014 the abstract describes an 'evaluation taxonomy with three suites' (fact-based QA, temporal reasoning, long-horizon decision-making) which is a graded, multi-dimensional structure, but no single explicit partial-credit formula is given.",
  "horizon_value": "19.72 inpatient visits per patient (average); 44.91 medical events per visit (average)",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per patient (episode = one patient's full longitudinal record)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "It comprises 335 patients, with 19.72 inpatient visits per patient on average and 44.91 medical events per visit.",
  "horizon_span": "It comprises 335 patients, with 19.72 inpatient visits per patient on average and 44.91 medical events per visit.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290118634",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "sessions",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "288652433",
  "title": "MBABench: Evaluating LLM Agents on End-to-End Spreadsheet Tasks in Finance",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-05-21",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "MBABench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "construct an entire financial spreadsheet (e.g., financial model, forecast, scenario analysis) end-to-end from a high-level instruction; jointly satisfy Accuracy, Formula, and Format criteria, each with fine-grained sub-criteria reflecting professional standards",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Because deliverables are routinely reviewed and revised by multiple stakeholders, readability/ease-of-modification (Format) interacts with correctness (Accuracy) and formula design (Formula); performance also degrades sharply as the number of chained calculations increases.",
  "n_goals": "three top-level evaluation dimensions (Accuracy, Formula, Format), each with fine-grained criteria; exact count of fine-grained criteria not given",
  "tracking_demand": "Agent must track intermediate calculation dependencies within the spreadsheet, professional formatting/readability conventions, and formula correctness simultaneously while building the deliverable end-to-end.",
  "scoring": "milestone-rubric - a three-dimension evaluation taxonomy (Accuracy, Formula, Format) with fine-grained criteria reflecting professional standards, evaluated across 18+ agents; this is explicit multi-criteria/rubric scoring rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Even the strongest agents fall short of basic professional finance standards, and performance degrades sharply as difficulty increases beyond a few chained calculations; no specific numeric top score or human baseline is given.",
  "availability": null,
  "goal_span": "To reflect the multidimensional nature of solution quality, we develop an evaluation taxonomy comprising three dimensions: Accuracy, Formula, and Format, each comprising fine-grained criteria that reflect professional standards.",
  "horizon_span": "Evaluating over 18 agents, the benchmark reveals that even the strongest agents fall short of basic professional finance standards, and their performance degrade sharply as the difficulty increases beyond a few chained calculations.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288652433",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290781442",
  "title": "MerchantBench: Benchmarking LLM Agents for Long-Term Coherence in E-Commerce Operations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-07-31",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "MerchantBench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "source products and manage upstream supplier events over time; set and adjust listing/pricing to remain competitive and solvent; manage cash-flow across delayed, heterogeneous-latency order outcomes; follow individual order lifecycles end-to-end and revisit earlier decisions as new (delayed) information arrives",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Product-sourcing decisions constrain what can be listed/priced, pricing/cash-flow decisions determine solvency, and delayed downstream order outcomes require revisiting and adapting earlier upstream decisions -- so an early sourcing or pricing mistake compounds via delayed feedback loops across the 365-day run.",
  "n_goals": "recurrent decisions across 4 named categories (Product Sourcing, Listing/Pricing Control, Cash-Flow Management, Mixed-Latency Feedback Adaptation), instantiated via a 365-day simulation grounded in 98,843 real product records with 26 interaction tools, across 48 runs",
  "tracking_demand": "The agent must follow individual order lifecycles end-to-end, track upstream supplier events and their promised delayed downstream outcomes, manage cash-flow/net-assets state, and revisit/adapt earlier sourcing and pricing decisions as delayed feedback arrives across a full 365-simulated-day run.",
  "scoring": "continuous-reward -- performance is measured via final net assets relative to human participants (best LLM configuration attains only 27.3% of the mean final net assets achieved by human participants); the abstract does not describe an explicit subgoal-checkpoint partial-credit rubric.",
  "horizon_value": "a 365-day order-level simulation; 48 runs, each spanning 365 simulated days",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/run (one full 365-simulated-day business-operation trajectory)",
  "horizon_stated": "yes",
  "headline_result": "Best LLM configuration attains only 27.3% of the mean final net assets achieved by human participants (evaluated across 8 LLMs, 2 agent frameworks, 48 runs of 365 simulated days each).",
  "availability": null,
  "goal_span": "Seller-side e-commerce provides a suitable setting for this evaluation through recurrent and interdependent decisions over Product Sourcing, Listing and Pricing Control, Cash-Flow Management, and Mixed-Latency Feedback Adaptation. We introduce MerchantBench, a 365-day order-level simulation grounded in 98,843 real e-commerce product records and equipped with 26 tools for agent interaction. MerchantBench couples promptly observable Upstream Supplier Events with delayed Downstream Order Outcomes, requiring agents to follow individual order lifecycles and revisit earlier decisions.",
  "horizon_span": "We introduce MerchantBench, a 365-day order-level simulation grounded in 98,843 real e-commerce product records and equipped with 26 tools for agent interaction... We evaluate eight LLMs under two agent frameworks in 48 runs, each spanning 365 simulated days.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290781442",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285615667",
  "title": "CORPGEN: Simulating Corporate Environments with Autonomous Digital Employees in Multi-Horizon Task Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-02-15",
  "months_since_pub": 7,
  "citations_per_month": 0.0,
  "artifact_name": "Multi-Horizon Task Environments (MHTE) / CorpGen",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "manage dozens of concurrent, interleaved long-horizon corporate tasks (45+ tasks, 500-1500+ steps each); handle inter-task dependencies expressed as DAGs rather than simple chains; reprioritize among concurrent tasks as load and context change over a persistent execution context spanning hours",
  "goal_origin": "mixed:given-up-front-with-reprioritization-emitted-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Concurrent tasks interleave and share execution context (memory, sub-agent isolation), dependencies form DAGs rather than simple chains, and reprioritization needs mean task ordering itself changes dynamically as load scales from 25% to 100%.",
  "n_goals": "45+ concurrent tasks, each requiring 500-1500+ steps",
  "tracking_demand": "The system must maintain hierarchical goal alignment, isolate sub-agent context to prevent cross-task contamination, and manage tiered (working/structured/semantic) memory with adaptive summarization across a persistent execution context spanning hours.",
  "scoring": "other:completion-rate-degradation-under-increasing-load",
  "horizon_value": "500-1500+",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (within a persistent multi-task execution context spanning hours)",
  "horizon_stated": "yes",
  "headline_result": "CorpGen achieves up to 3.5x improvement over baselines (15.2% vs 4.3%) across three CUA backends (UFO2, OpenAI CUA, hierarchical) on OSWorld Office, with stable performance under increasing load; no human baseline reported.",
  "availability": null,
  "goal_span": "We identify four failure modes that cause baseline CUAs to degrade from 16.7% to 8.7% completion as load scales 25% to 100%, a pattern consistent across three independent implementations. These failure modes are context saturation (O(N) vs O(1) growth), memory interference, dependency complexity (DAGs vs. chains), and reprioritization overhead.",
  "horizon_span": "requiring coherent execution across dozens of interleaved tasks (45+, 500-1500+ steps) within persistent execution contexts spanning hours.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285615667",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288702065",
  "title": "OR-Space: A Full-Lifecycle Workspace Benchmark for Industrial Optimization Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-05-27",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "OR-Space",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "construct a solver-ready optimization model from heterogeneous business artifacts (Build); revise an existing model under changing requirements or solver feedback while preserving valid prior logic (Revise); answer grounded questions about solutions, constraints, and business implications using evidence spread across workspace artifacts (Explain)",
  "goal_origin": "mixed:given-up-front-with-new-requirements-injected-during-revise",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Model construction, revision, and explanation all read and write a shared set of interdependent workspace files (business documents, structured data, code, solver outputs), so a revision must preserve valid prior logic while satisfying new requirements, and explanations must stay grounded in the current artifact state.",
  "n_goals": "3 task modes (Build, Revise, Explain)",
  "tracking_demand": "The agent must track the current state of a persistent multi-artifact workspace (documents, structured data, code, solver outputs) across model construction, revision, and explanation stages, preserving valid prior modeling logic when new requirements or solver feedback arrive.",
  "scoring": "other:not-specified-in-abstract",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "OR-Space defines three task modes: Build, where agents construct solver-ready optimization models from heterogeneous artifacts; Revise, where agents modify existing models under changing requirements or solver feedback while preserving valid prior logic; and Explain, where agents answer grounded questions about solutions, constraints, and business implications using evidence spread across workspace artifacts.",
  "horizon_span": "persistent multi-artifact workspaces and multi-stage task lifecycles",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288702065",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "291183436",
  "title": "UI-Mate: Advancing Open-Weight Foundation GUI Agents with In-Context Demonstrations",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-16",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "OSWorkerBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 100 long-horizon office tasks spanning 41 applications; follow demonstrated subtask-level workflows (self-demo or variant-demo) and re-plan from the live interface when needed; make measurable progress toward task completion even when strict end-to-end success is not reached",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Demonstrations are transformed into subtask-level workflows the agent follows and re-plans from as the live interface diverges from the demonstrated steps, so later subtask execution depends on correctly tracking progress against (and deviations from) earlier demonstrated or self-derived steps.",
  "n_goals": "100 long-horizon office tasks across 41 applications; split into 33 self-demo and 45 variant-demo tasks",
  "tracking_demand": "Agent must track its position in a demonstrated or self-planned subtask sequence, detect when the live interface diverges from the demonstration, and re-plan accordingly across a long-horizon office task.",
  "scoring": "subgoal-checkpoint-partial-credit - reports both strict success and progress metrics (e.g., 41.0% strict success and 76.9% progress), where progress functions as explicit partial credit toward complete task success, distinct from a purely binary outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "UI-Mate-27B reaches 41.0% strict success and 76.9% progress on OSWorkerBench, outperforming its Qwen3.6-27B base by 17.7 and 24.5 points; on the 33-task self-demo subset, one demonstration raises strict success from 17.2% to 35.4% and progress from 67.9% to 81.1%; no human/expert baseline is reported for OSWorkerBench itself.",
  "availability": "https://ui-mate.github.io",
  "goal_span": "OSWorkerBench Benchmark and Insights: A benchmark of 100 long-horizon office tasks across 41 applications that supports instruction-only and demonstration-guided evaluation.",
  "horizon_span": "OSWorkerBench Benchmark and Insights: A benchmark of 100 long-horizon office tasks across 41 applications that supports instruction-only and demonstration-guided evaluation.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291183436",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287432902",
  "title": "OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 4,
  "publication_date": "2026-04-13",
  "months_since_pub": 5,
  "citations_per_month": 0.8,
  "artifact_name": "OccuBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 100 real-world professional task scenarios spanning 65 specialized domains; maintain task completion under controlled fault injection (explicit errors, implicit data degradation, mixed faults)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Within a scenario, the agent must detect and adapt to injected faults (explicit errors, degraded data, or both) that alter the reliability of tool responses mid-task, but the 100 scenarios are otherwise independent of each other.",
  "n_goals": "100 real-world professional task scenarios across 10 industry categories and 65 specialized domains",
  "tracking_demand": "Agent must track task-completion progress plus signals of environmental robustness (whether tool responses are timing out, truncated, or subtly degraded) within each professional scenario.",
  "scoring": "other:task-completion-plus-robustness-score - evaluated along two complementary dimensions (task completion, environmental robustness under fault injection); abstract does not describe fine-grained subgoal-checkpoint credit within a single scenario.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "GPT-5.2 improves by 27.5 points from minimal to maximum reasoning effort; no single model dominates across all industries; no explicit human/expert comparison is given.",
  "availability": null,
  "goal_span": "OccuBench evaluates agents along two complementary dimensions: task completion across professional domains and environmental robustness under controlled fault injection (explicit errors, implicit data degradation, and mixed faults).",
  "horizon_span": "We introduce OccuBench, a benchmark covering 100 real-world professional task scenarios across 10 industry categories and 65 specialized domains, enabled by Language Environment Simulators (LESs) that simulate domain-specific environments through LLM-driven tool response generation.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287432902",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290648033",
  "title": "OmegaUse-OfficeVal: Benchmarking LLM Agents on Long-Horizon Office-Suite Tasks with Economic Grounding",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-29",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "OmegaUse-OfficeVal",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete 100 practitioner-derived office-suite tasks end-to-end; achieve deliverable quality verified via fine-grained rubric-based code verifiers; remain economically competitive relative to human labor time and task price",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each of the 100 tasks is an independent office-suite request, but all are jointly compared via shared economic signals (human labor time, task price proxy) enabling value-weighted evaluation across the whole set.",
  "n_goals": "100 tasks derived from practitioner office-suite requests",
  "tracking_demand": "Agent must track fine-grained rubric criteria and produce a deliverable matching the practitioner's request, verified via code-based verifiers, implicitly benchmarked against the human labor time (avg. 2.32 hours) needed for the same task.",
  "scoring": "milestone-rubric",
  "horizon_value": "2.32 (average)",
  "horizon_unit": "human-expert-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "All evaluated LLMs are substantially cheaper and faster than the human baseline but have not yet approached human-level deliverable quality; average human labor time per task is 2.32 hours (no single numeric success-rate gap given in the abstract).",
  "availability": null,
  "goal_span": "To support stable evaluation, we develop code-based verifiers from fine-grained rubrics. We evaluate several frontier LLMs together with a human baseline. Although all evaluated LLMs are substantially cheaper and faster than human workers, they have not yet approached human-level deliverable quality.",
  "horizon_span": "The benchmark comprises 100 tasks derived from office-suite requests proposed by practitioners and adapted through a privacy-preserving process. On average, these tasks require 2.32 hours of human labor to complete.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290648033",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "human-expert-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285102407",
  "title": "OptAgent: an Agentic AI framework for Intelligent Building Operations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:building-energy-operations",
  "citation_count": 1,
  "publication_date": "2026-01-27",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "OptAgent",
  "artifact_kind": "benchmark",
  "domain": "other:building-energy-operations",
  "goal_types": "assess how a system/control upgrade changes energy use; assess how the same upgrade changes operating cost; assess how the same upgrade changes thermal comfort; assess how the same upgrade changes flexibility, via coordinated multi-domain multi-agent analytics",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A single case study requires cross-domain coordination among 11 specialist agents and 72 MCP tools, where thermal-dynamics, HVAC, and DER outputs feed into each other, so the energy-use, cost, comfort, and flexibility assessments are computed from a shared, dependency-ordered simulation/analytics pipeline rather than independently.",
  "n_goals": "large-scale benchmark of about 4,000 runs; 11 specialist agents; 72 MCP tools",
  "tracking_demand": "The orchestrator must track which specialist agent/tool has been invoked for which sub-domain (thermal dynamics, HVAC, DER), intermediate physics-informed simulation outputs, and how upgrades propagate across energy use, cost, comfort, and flexibility metrics within one workflow.",
  "scoring": "other:accuracy-token-consumption-execution-time-cost",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "an agentic AI layer with 11 specialist agents and 72 Model Context Protocol (MCP) tools that enable end-to-end execution of multi-step energy analytics. A representative case study demonstrates multi-domain, multi-agent coordination for assessing how system and control upgrades affect energy use, operating cost, thermal comfort, and flexibility.",
  "horizon_span": "a large-scale benchmark (about 4000 runs) systematically evaluates workflow performance in terms of accuracy, token consumption, execution time, and inference cost",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285102407",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290149026",
  "title": "Evidence-Grounded AI for Musculoskeletal Care",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-14",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "OrthoPilot",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "integrate evolving imaging, laboratory, pathology, and order data as it arrives across visits; produce evidence-based decisions at each stage of care, from admission diagnosis through rehabilitation planning; maintain continuous, individualised management across the full musculoskeletal care pathway rather than isolated per-visit decisions",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later-stage decisions (e.g. rehabilitation planning) depend on evidence accumulated and integrated at earlier stages (e.g. admission diagnosis), since recovery/degeneration 'unfold over months to years' and care 'requires longitudinal management rather than isolated decisions'.",
  "n_goals": null,
  "tracking_demand": "System must continuously retrieve and integrate real-time imaging, laboratory, pathology, and order data across visits/departments/hospital systems, translating evolving patient state into stage-specific functional goals across the whole care pathway.",
  "scoring": "other:mixed \u2014 includes a specialist-validated benchmark (1,000 disease codes) with expert comparison, a prospective decision-making study (full-chain management success rate, +10.6%), and a randomised deployment study (+9.7% cases per bed); this is closer to milestone-rubric evaluated by human experts than binary success.",
  "horizon_value": "months to years (per patient pathway); 1,870 cases / 8,240 inpatients across the study",
  "horizon_unit": "other:real-world-months-to-years-per-patient-pathway",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per patient (one full care pathway/episode spans months to years of real time, aggregated across thousands of patients in the studies)",
  "horizon_stated": "yes",
  "headline_result": "OrthoPilot outperformed 81 orthopaedic physicians with up to 25 years of experience in a full-pathway reader study, generalised across 60 external centres, improved full-chain management success by 10.6% in a prospective study of 1,870 cases, and increased cumulative cases per bed by 9.7% in a randomised deployment across 8,240 inpatients.",
  "availability": null,
  "goal_span": "Clinicians must repeatedly integrate evolving patient evidence, medical knowledge and stage-specific functional goals, yet evidence is often fragmented across visits, departments and hospital systems, disrupting continuous, individualised management.",
  "horizon_span": "recovery, remodelling and degeneration of bones, joints and related tissues unfold over months to years, care requires longitudinal management rather than isolated decisions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": "This reads as a deployed clinical decision-support system evaluated against physicians in real hospitals (reader studies, randomised deployment) rather than a typical LLM-agent benchmark/environment with self-contained episodes; may be a better fit for a healthcare-AI corpus than a long-horizon agentic-benchmark corpus.",
  "url": "https://api.semanticscholar.org/CorpusId:290149026",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284912111",
  "title": "POLARIS: Typed Planning and Governed Execution for Agentic AI in Back-Office Automation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-01-16",
  "months_since_pub": 8,
  "citations_per_month": 0.25,
  "artifact_name": "POLARIS",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "synthesize a type-checked directed acyclic graph (DAG) plan for a back-office document-processing task; select a single compliant plan via rubric-guided reasoning among structurally diverse candidate DAGs; pass validator-gated checks and a bounded repair loop before execution; route or block side effects per compiled policy guardrails, including anomaly routing",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The typed plan is itself a DAG, so plan nodes carry explicit precedence and type constraints, and validator-gated checks plus policy guardrails block downstream execution steps whose upstream nodes fail or violate policy.",
  "n_goals": null,
  "tracking_demand": "System must track the candidate DAG structure, per-node type/validator status, and the full execution trace/audit trail to support bounded repair and decision-grade anomaly routing.",
  "scoring": "other:micro-f1-plus-precision-on-synthetic-suite - reports micro F1 (0.81) on SROIE and 0.95-1.00 precision on anomaly routing on a controlled synthetic suite; abstract does not describe a formal subgoal-checkpoint partial-credit scheme beyond these end metrics, though validator-gated checks act as implicit per-node gating.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "POLARIS achieves micro F1 of 0.81 on the SROIE dataset and 0.95 to 1.00 precision for anomaly routing on a controlled synthetic suite; no human/expert baseline is given.",
  "availability": null,
  "goal_span": "A planner proposes structurally diverse, type checked directed acyclic graphs (DAGs), a rubric guided reasoning module selects a single compliant plan, and execution is guarded by validator gated checks, a bounded repair loop, and compiled policy guardrails that block or route side effects before they occur.",
  "horizon_span": "Applied to document centric finance tasks, POLARIS produces decision grade artifacts and full execution traces while reducing human intervention.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": "POLARIS is presented primarily as a governed orchestration framework; the paper itself describes its evaluations as only 'an initial benchmark for governed Agentic AI', so the artifact is closer to a framework-plus-pilot-evaluation than an established benchmark suite.",
  "url": "https://api.semanticscholar.org/CorpusId:284912111",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289691134",
  "title": "PPT-Eval: A Benchmark for Computer-Use Agents on PowerPoint Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 5,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 1.67,
  "artifact_name": "PPT-Eval",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete content-creation and presentation-editing tasks across 120 PowerPoint tasks in 12 files; satisfy task-specific rubric criteria that award partial credit for intermediate steps; avoid unnecessary changes and poor aesthetics while making required edits",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Rubrics award partial credit for intermediate steps and penalize unnecessary changes, so an agent's overall score depends on correctly completing each rubric-relevant sub-step without introducing side effects elsewhere in the presentation.",
  "n_goals": "120 PowerPoint tasks across 12 files, organized by difficulty",
  "tracking_demand": "Agent must track which task-specific rubric criteria (intermediate steps, aesthetics, unnecessary changes) have been satisfied across a multimodal editing session, since rubrics award partial credit rather than only final binary success.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Claude-4.5-Opus achieves only a 45% binary success rate but an average partial score of 57%; the rubric evaluator correlates with human judgments at Kendall's tau-b = 0.77; no human baseline directly compared on task success.",
  "availability": "https://microsoft.github.io/ppteval",
  "goal_span": "we design a robust evaluation framework to help create task-specific rubrics for PowerPoint tasks... These rubrics award partial credit for intermediate steps, penalize unnecessary changes and poor aesthetics, and provide natural language feedback. This nuanced approach proves highly effective, achieving a Kendall's tau-b correlation of 0.77 with human judgments.",
  "horizon_span": "We introduce PPT-Eval, a benchmark of 120 PowerPoint tasks across 12 files that cover both content creation and presentation editing scenarios, organized by difficulty.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289691134",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287950837",
  "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 14,
  "publication_date": "2026-05-04",
  "months_since_pub": 4,
  "citations_per_month": 3.5,
  "artifact_name": "PhysicianBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "retrieve relevant clinical data across multiple encounters in the EHR; reason over heterogeneous clinical information (labs, notes, orders) to reach a decision; execute consequential clinical actions (e.g. prescribing, ordering) grounded against the environment; produce clinical documentation reflecting the completed workflow",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each task is decomposed into structured checkpoints capturing distinct stages of completion, so later stages (e.g. treatment planning, documentation) depend on correctly retrieving and reasoning over data gathered at earlier stages/encounters.",
  "n_goals": "100 long-horizon tasks; 670 checkpoints total across the benchmark (~6.7 checkpoints/task on average)",
  "tracking_demand": "Agent must track data retrieved across multiple encounters, intermediate clinical reasoning state, and which structured checkpoints have been satisfied, using execution-grounded verification against real patient records via standard EHR APIs.",
  "scoring": "subgoal-checkpoint-partial-credit \u2014 explicitly, 'Each task is decomposed into structured checkpoints (670 in total across the benchmark) capturing distinct stages of completion graded by task-specific scripts with execution-grounded verification.'",
  "horizon_value": "27 (average)",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Best-performing model achieves only 46% success rate (pass@1) across 13 proprietary and open-source LLM agents, while open-source models reach at most 19% (no human-physician baseline given, though tasks were physician-reviewed).",
  "availability": null,
  "goal_span": "Each task is decomposed into structured checkpoints (670 in total across the benchmark) capturing distinct stages of completion graded by task-specific scripts with execution-grounded verification.",
  "horizon_span": "requiring an average of 27 tool calls per task",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287950837",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289941211",
  "title": "PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": null,
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "PolyWorkBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "process heterogeneous multilingual inputs correctly within a workflow task; perform iterative reasoning and invoke external tools while maintaining linguistic consistency; produce a structured, correct output for one of five workplace domains (commerce, knowledge work, legal analysis, localization, manufacturing)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The abstract states 'multilinguality introduces compounding effects across reasoning and execution steps', meaning language-consistency errors at one step in the workflow can propagate and compound into later reasoning/tool-invocation steps of the same task.",
  "n_goals": "67 tasks across 5 domains",
  "tracking_demand": "Agent must track functional correctness and linguistic consistency simultaneously across a workflow's reasoning and tool-invocation steps, since the two can drift independently as multilingual inputs are processed.",
  "scoring": "other:mixed \u2014 'a hybrid framework that combines structural grading, executable verification, and LLM-based semantic assessment... to capture both functional correctness and linguistic consistency', i.e. multiple graded dimensions rather than a single binary outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "State-of-the-art LLM agents suffer significant performance degradation in multilingual workflow settings compared to monolingual counterparts (no single numeric headline score or human baseline given).",
  "availability": null,
  "goal_span": "PolyWorkBench consists of 67 tasks across five domains, including commerce, knowledge work, legal analysis, localization, and manufacturing, where agents must process heterogeneous multilingual inputs, perform iterative reasoning, invoke external tools, and produce structured outputs.",
  "horizon_span": "PolyWorkBench consists of 67 tasks across five domains",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289941211",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291182836",
  "title": "PolyWorkBench: Benchmarking LLM Agents for Cross-Lingual Long-Horizon Workflows",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-07-07",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "PolyWorkBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "integrate heterogeneous multilingual inputs relevant to a workplace task; execute iterative tool-use trajectories within a target domain (commerce, knowledge work, legal analysis, localization, manufacturing); produce structured domain artifacts as verifiable task output",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Correct integration of multilingual inputs constrains later structured-artifact production, and multilingual execution is shown to expose failure modes cascading across planning, tool interaction, and decision-making stages.",
  "n_goals": "67 tasks across five core domains: commerce, knowledge work, legal analysis, localization, manufacturing",
  "tracking_demand": "Agent must track and correctly integrate heterogeneous multilingual information while executing an iterative sequence of tool calls, and produce a structured domain artifact that is later verified.",
  "scoring": "other:grade-structural-rubric-plus-pytest-plus-llm-judge",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we adopt Grade, a task-specific structural scoring rubric, as our primary ranking metric, and complement it with Pytest for executable state verification and LLM-as-Judge for semantic quality diagnostics.",
  "horizon_span": "enterprise workflows inherently require processing multilingual resources across extended trajectories. The interaction between multilinguality and long-horizon execution, however, remains underexplored.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291182836",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289630434",
  "title": "PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:power-systems-engineering",
  "citation_count": 4,
  "publication_date": "2026-06-17",
  "months_since_pub": 3,
  "citations_per_month": 1.33,
  "artifact_name": "PowerAgentBench-SS",
  "artifact_kind": "benchmark",
  "domain": "other:power-systems-engineering",
  "goal_types": "inspect a grid case and select appropriate tools/simulators for the workflow; screen a large space of contingencies within a limited validation budget; propose admissible mitigations for discovered risks and validate their physical validity; produce an auditable evidence trail and submit a bounded, ranked report of top contingencies",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The agent has a shared, limited validation budget (80 validations) that it must allocate across screening 1,035 total N-2 contingency cases to find the hidden set of 52 dangerous cases, so validating one candidate contingency uses up budget that could have gone toward another, and the final ranked report of exactly 20 contingencies must reflect the evidence accumulated from those budgeted validations.",
  "n_goals": "1,035 total N-2 contingency cases per instance; validation budget B=80; hidden dangerous set of 52 cases (top 5% by severity); submitted report size m=20",
  "tracking_demand": "The agent must track which contingencies it has already validated (against a fixed budget of 80 out of 1,035), the evidence log supporting each, and whether its running set of top candidates still reflects the best available evidence as it allocates its remaining validation budget.",
  "scoring": "other:risk-sensitive-multi-metric-scoring. The paper defines multiple risk-sensitive metrics -- submitted recall, evidence-backed recall, found recall, false-safe penalties, severity regret, residual violation score, action cost, tool-use efficiency, and workflow diagnostics -- explicit multi-dimensional, process-aware scoring rather than a single pass/fail or milestone rubric.",
  "horizon_value": "validation budget of 80 contingency-case validations per episode (out of 1,035 total N-2 cases), submitting a ranked report of exactly 20 contingencies; hidden dangerous set of 52 cases",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per instance of the contingency-search pilot)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "The benchmark exposes public case data, action constraints, a tool API, and a validation budget to an agent, while a hidden evaluator recomputes physical validity and scores the submitted report. We define the agent interface, tool contract, evidence log, and risk-sensitive metrics, including submitted recall, evidence-backed recall, found recall, false-safe penalties, severity regret, residual violation score, action cost, tool-use efficiency, and workflow diagnostics.",
  "horizon_span": "The validation budget is B=80 ... The report size is m=20 ... the top 5% of cases by severity",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289630434",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286583757",
  "title": "RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 6,
  "publication_date": "2026-03-17",
  "months_since_pub": 6,
  "citations_per_month": 1.0,
  "artifact_name": "RetailBench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "manage pricing across the store's product assortment; manage replenishment and supplier selection to keep inventory stocked; manage shelf assortment and inventory aging; respond appropriately to customer feedback and external events; maintain solvency (cash-flow constraints) while maximizing net worth/sales over a long simulated horizon",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "RetailBench 'models retail management as a partially observable decision process,' so pricing, replenishment, and assortment decisions interact through shared inventory/cash-flow state across the run, and the paper attributes performance gaps to 'incomplete evidence acquisition' and 'lack of a consistent long-horizon policy.'",
  "n_goals": "7 contemporary LLMs evaluated under representative agent frameworks over a 180-day horizon; compared to a privileged oracle policy",
  "tracking_demand": "The agent must track pricing, inventory levels/aging, supplier relationships, customer feedback, external events, and its own cash-flow position day by day across the 180-day (or longer) simulated run, since only a small subset of evaluated agents survive the full evaluation horizon.",
  "scoring": "other:final-net-worth-and-sales-vs-oracle. Agents are compared to 'a privileged oracle policy' on 'final net worth and sales outcomes,' a continuous, cumulative outcome metric benchmarked against a strong reference policy rather than a milestone/subgoal-checkpoint rubric.",
  "horizon_value": "180-day evaluation horizon; the simulator supports thousand-day-scale simulations",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole simulated store-management run)",
  "horizon_stated": "yes",
  "headline_result": "Only a small subset of the 7 evaluated LLMs survive the full 180-day evaluation horizon, and even the strongest LLM runs remain substantially behind a privileged oracle policy in final net worth and sales outcomes; no human/expert baseline given.",
  "availability": null,
  "goal_span": "RetailBench models retail management as a partially observable decision process and is designed to support thousand-day-scale simulations. In this environment, agents must manage pricing, replenishment, supplier selection, shelf assortment, inventory aging, customer feedback, external events, and cash-flow constraints. We evaluate seven contemporary LLMs under representative agent frameworks over a 180-day evaluation horizon and compare them with a privileged oracle policy.",
  "horizon_span": "We evaluate seven contemporary LLMs under representative agent frameworks over a 180-day evaluation horizon and compare them with a privileged oracle policy... RetailBench...is designed to support thousand-day-scale simulations.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286583757",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287479679",
  "title": "RiskWebWorld: A Realistic Interactive Benchmark for GUI Agents in E-commerce Risk Management",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-04-15",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "RiskWebWorld",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "investigate flagged e-commerce risk cases across multiple verification sub-steps on production risk-control pipelines; operate GUI actions on uncooperative websites subject to partial environmental hijacking; complete each of 1,513 tasks spanning 8 core risk-control domains; succeed at long-horizon professional risk-investigation workflows",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Risk investigation tasks require chaining multiple verification sub-steps across an uncooperative, sometimes environmentally-hijacked website, so later verification steps depend on correctly completing and interpreting earlier steps within the same case.",
  "n_goals": "1,513 tasks sourced from production risk-control pipelines across 8 core domains",
  "tracking_demand": "The agent must track evidence and verification state gathered across multiple sub-steps of a risk investigation on an uncooperative website, while detecting and coping with partial environmental hijacking attempts, across long-horizon professional tasks.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Top-tier generalist models achieve 49.1% success on RiskWebWorld, while specialized open-weight GUI models are near-total failure; agentic RL improves open-source models by 16.2%. No explicit human/expert baseline is given.",
  "availability": null,
  "goal_span": "Our evaluation across diverse models reveals a dramatic capability gap: top-tier generalist models achieve 49.1% success, while specialized open-weights GUI models lag at near-total failure.",
  "horizon_span": "This highlights that foundation model scale currently matters more than zero-shot interface grounding in long-horizon professional tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287479679",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288392942",
  "title": "SaaS-Bench: Can Computer-Use Agents Leverage Real-World SaaS to Solve Professional Workflows?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 3,
  "publication_date": "2026-05-15",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "SaaS-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "navigate and operate real, deployed SaaS systems to complete a professional workflow; coordinate state and context across multiple applications within the same workflow; apply domain-specific knowledge correctly within the SaaS system; recover from errors and maintain progress over a long-horizon task",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks 'naturally involve dynamic system states, cross-application coordination, domain-specific knowledge, and long-horizon dependencies,' so completing a later step in one SaaS application typically depends on state changes made in an earlier step, possibly in a different application, and weighted verification checkpoints track partial progress toward the final goal rather than only a final state check.",
  "n_goals": "106 tasks grounded in realistic work scenarios, built on 23 deployable SaaS systems across 6 professional domains",
  "tracking_demand": "The agent must maintain state and context across multiple SaaS applications over long-horizon execution, tracking partial progress against weighted verification checkpoints, since 'agents become stuck acquiring information or manipulating interfaces, confuse source and output devices, or terminate before all conditions are jointly satisfied.'",
  "scoring": "subgoal-checkpoint-partial-credit. Tasks 'are evaluated with weighted verification checkpoints that measure strict task completion and partial progress,' explicit partial credit for intermediate checkpoints, not just a final binary success flag.",
  "horizon_value": "average of over 100 interaction steps per task",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Even the strongest evaluated model completes fewer than 4% of tasks end-to-end on SaaS-Bench; no human/expert baseline given in the abstract.",
  "availability": null,
  "goal_span": "SaaS-Bench, a benchmark built on 23 deployable SaaS systems across six professional domains, containing 106 tasks grounded in realistic work scenarios. These tasks require long-horizon execution, cover both text-only and multimodal settings, and are evaluated with weighted verification checkpoints that measure strict task completion and partial progress.",
  "horizon_span": "SaaS-Bench introduces long-horizon tasks with an average of over 100 interaction steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288392942",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289684931",
  "title": "SpreadsheetBench 2: Evaluating Agents on End-to-End Business Spreadsheet Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 4,
  "publication_date": "2026-06-29",
  "months_since_pub": 3,
  "citations_per_month": 1.33,
  "artifact_name": "SpreadsheetBench 2",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "generate new spreadsheet content/formulas correctly across a large multi-sheet workbook; debug existing incorrect formulas/content within the workbook; produce correct visualizations from the workbook's data",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Cross-sheet dependencies mean a cell modification on one worksheet can be required by or affect formulas/values on another worksheet within the same workbook, so correctness depends on consistently propagating changes across an average of 11.8 worksheets and 593.5 cell modifications.",
  "n_goals": "321 tasks; each instance averages 11.8 worksheets and requires 593.5 cell modifications, across three task categories (generation, debugging, visualization)",
  "tracking_demand": "The agent must track and correctly propagate changes across an average of 11.8 interdependent worksheets requiring 593.5 cell modifications per task, correctly identifying target cells under a unified multi-turn agent scaffold.",
  "scoring": "binary-final-success -- the best model achieves 34.89% overall task accuracy, with debugging accuracy as low as 12.00%, framed as per-task accuracy; the abstract does not describe intra-task partial credit beyond overall task-level accuracy per category.",
  "horizon_value": "each task instance averages 11.8 worksheets and requires 593.5 cell modifications",
  "horizon_unit": "other:cell-modifications",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Best model achieves 34.89% overall task accuracy, with debugging accuracy as low as 12.00%; no human/expert baseline is reported.",
  "availability": "https://spreadsheetbench.github.io/",
  "goal_span": "The benchmark contains 321 tasks; each instance averages 11.8 worksheets and requires 593.5 cell modifications, reflecting large multi-sheet workbooks with cross-sheet dependencies.",
  "horizon_span": "The benchmark contains 321 tasks; each instance averages 11.8 worksheets and requires 593.5 cell modifications, reflecting large multi-sheet workbooks with cross-sheet dependencies.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289684931",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285451644",
  "title": "SupChain-Bench: Benchmarking Large Language Models for Real-World Supply Chain Management",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-02-07",
  "months_since_pub": 7,
  "citations_per_month": 0.0,
  "artifact_name": "SupChain-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "orchestrate long-horizon, multi-step supply-chain tool use grounded in standard operating procedures (SOPs); correctly apply supply-chain domain knowledge across a sequence of dependent tool calls",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "SOP-grounded orchestration requires multi-step tool calls to be executed in the procedurally correct order, so execution reliability depends on correctly sequencing dependent tool calls per the governing SOP.",
  "n_goals": null,
  "tracking_demand": "Agent must track SOP-grounded procedural state across a long-horizon sequence of dependent tool calls in order to correctly complete supply-chain management workflows.",
  "scoring": "other:tool-calling-performance-comparison",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "supply chain workflows require reliable long-horizon, multi-step orchestration grounded in domain-specific procedures, which remains challenging for current models... we introduce SupChain-Bench, a unified real-world benchmark that assesses both supply chain domain knowledge and long-horizon tool-based orchestration grounded in standard operating procedures (SOPs).",
  "horizon_span": "supply chain workflows require reliable long-horizon, multi-step orchestration grounded in domain-specific procedures, which remains challenging for current models.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285451644",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287915547",
  "title": "Synthetic Computers at Scale for Long-Horizon Productivity Simulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 1,
  "publication_date": "2026-04-30",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "Synthetic Computers at Scale",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of multiple professional deliverables comprising a computer-specific productivity objective; navigate the synthetic computer's filesystem to ground actions in the user's actual context; coordinate with simulated collaborators as needed to complete the objective; sustain progress toward an objective requiring about a month of simulated human work",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "One agent generates the productivity objectives specific to a synthetic computer's user, and a second agent must keep working across that same computer -- navigating the filesystem, coordinating with collaborators, and producing artifacts -- such that later deliverables depend on context (folder hierarchies, prior artifacts) established earlier in the same long run.",
  "n_goals": "1,000 synthetic computers; each run requires >8 hours of agent runtime and >2,000 turns on average",
  "tracking_demand": "The acting agent must track the evolving state of a realistic folder hierarchy and content-rich artifacts (documents, spreadsheets, presentations) across a run spanning over 2,000 turns, coordinating with simulated collaborators toward multiple professional deliverables.",
  "scoring": "other:downstream-in-domain-and-out-of-domain-performance-improvement",
  "horizon_value": ">2,000 turns on average (each run also requiring over 8 hours of agent runtime)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run",
  "horizon_stated": "yes",
  "headline_result": "Significant improvements in agent performance on both in-domain and out-of-domain productivity evaluations from the resulting experiential learning signals; no single numeric headline given.",
  "availability": null,
  "goal_span": "one agent creates productivity objectives that are specific to the computer's user and require multiple professional deliverables and about a month of human work; another agent then acts as that user and keeps working across the computer -- for example, navigating the filesystem for grounding, coordinating with simulated collaborators, and producing professional artifacts -- until these objectives are completed. ... each run requires over 8 hours of agent runtime and spans more than 2,000 turns on average.",
  "horizon_span": "each run requires over 8 hours of agent runtime and spans more than 2,000 turns on average.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287915547",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "turns",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291257563",
  "title": "One Success Isn't Reliability: Thinkingbox, a Sandbox and Benchmark for Agents in Stateful Business Workflows",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-20",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Thinkingbox-bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "gather missing information over multiple turns before acting; follow domain-specific policies for each of 507 policy-conditioned workflows; coordinate dependent tools correctly; realize exactly the correct persistent backend state transition without collateral effects",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Correct completion requires coordinating dependent tools and realizing exactly the correct final state transition; wrong, missing, or extra effects are explicitly rejected by the checker, so a single erroneous action anywhere in the process invalidates state-transition correctness.",
  "n_goals": "507 policy-conditioned workflows across retail, hospitality, auto insurance, neobank internal IT, and consulting IT/HR support",
  "tracking_demand": "Agent must gather missing information across multiple turns, track applicable domain policies, coordinate dependent tool calls, and verify the resulting persistent backend state matches exactly the required final state with no collateral effects.",
  "scoring": "other:executable-checker-plus-pass@k",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The strongest model, Claude Opus 5, achieves 66.50% pass@1 but only 47.53% pass^20, showing a large gap between occasional success and reliable completion; no human baseline reported.",
  "availability": "https://github.com/microsoft/thinkingbox",
  "goal_span": "Each attempt is evaluated by task-specific executable checks that accept valid trajectories while rejecting wrong, missing, or extra effects; designated tasks additionally check required properties of the final response... the strongest model Claude Opus 5 achieves 66.50% pass@1, but only 47.53% pass^20.",
  "horizon_span": "Thinkingbox-bench contains 507 policy-conditioned workflows across numerous scenarios, including retail, hospitality, auto insurance, neobank internal IT, and consulting IT/HR support.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291257563",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285269064",
  "title": "Benchmarking Agents in Insurance Underwriting Environments",
  "year": 2026,
  "venue": "CAIS",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-01-31",
  "months_since_pub": 8,
  "citations_per_month": 0.0,
  "artifact_name": "Underwrite",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "gather information carefully from noisy tool interfaces and imperfect simulated users during an underwriting conversation; apply proprietary business/domain knowledge correctly rather than hallucinating it; reach a final underwriting decision consistent with the accumulated evidence gathered across the conversation",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Reaching a correct final underwriting decision depends on the specific evidence and clarifications gathered turn-by-turn from noisy tools and an imperfect simulated user earlier in the conversation, so the agent's information-gathering choices constrain what decision it can defensibly reach later.",
  "n_goals": null,
  "tracking_demand": "The agent must track what information it has already gathered (and from where), the reliability of that information given noisy tool interfaces, and how it should update its evolving underwriting assessment as more evidence accumulates across the conversation.",
  "scoring": "other:multi-model-accuracy-with-pass-at-k-consistency. The paper evaluates 13 frontier models and reports 'pass@k results show a 20% drop in performance,' consistency-across-attempts is explicitly measured in addition to raw accuracy, though no formal subgoal-checkpoint rubric is described in the abstract.",
  "horizon_value": "average of 3-7 steps of required reasoning and tool use, with a total of 10-20 conversational turns",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per underwriting conversation)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We present Underwrite, an expert-first, multi-turn insurance underwriting benchmark designed in close collaboration with domain experts to capture real-world enterprise challenges. Underwrite introduces critical realism factors often absent in current benchmarks: proprietary business knowledge, noisy tool interfaces, and imperfect simulated users requiring careful information gathering.",
  "horizon_span": "We aimed for an average of 3-7 steps of required reasoning and tool use, with a total of 10-20 conversational turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285269064",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289132980",
  "title": "Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2026-06-09",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "Workflow-GYM",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "autonomously operate domain-specific professional software GUIs to accomplish economically valuable work; complete long-horizon, multi-stage professional workflows end-to-end; maintain workflow consistency across stages without omission, error propagation, or objective drift",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Professional workflows unfold across multiple stages, and the paper finds agents suffer 'workflow stage omission, error propagation, and objective drift,' indicating each stage's output constrains subsequent stages, so an error or omission early in the workflow compounds through later stages.",
  "n_goals": null,
  "tracking_demand": "The agent must track its position and completed stages within a long-horizon, multi-stage professional GUI workflow, avoiding stage omission, propagating errors, or drifting from the original objective across the workflow's duration.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the strongest state-of-the-art models achieve only slightly above 30% success rate on Workflow-GYM's professional long-horizon GUI workflows; no explicit human/expert baseline percentage is given in the abstract.",
  "availability": null,
  "goal_span": "Through extensive experiments on state-of-the-art models, we find that even the strongest models achieve only slightly above 30% success rates, highlighting that professional long-horizon GUI workflows remain highly challenging for current GUI agents.",
  "horizon_span": "existing benchmarks rarely evaluate whether agents can operate graphical user interfaces to complete long-horizon, high-value professional workflows across diverse domains",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289132980",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287962808",
  "title": "Workspace-Bench 1.0: Benchmarking AI Agents on Workspace Tasks with Large-Scale File Dependencies",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 10,
  "publication_date": "2026-05-05",
  "months_since_pub": 4,
  "citations_per_month": 2.5,
  "artifact_name": "Workspace-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "identify, reason over, exploit, and update explicit and implicit dependencies among heterogeneous files in a worker's workspace; satisfy each of a task's own file-dependency-graph-derived rubrics via cross-file retrieval, contextual reasoning, and adaptive decision-making",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each of the 388 tasks has its own file dependency graph, so completing the task correctly requires respecting the ordering/dependency of files (e.g., a spreadsheet referencing an earlier report) rather than treating files independently.",
  "n_goals": "388 tasks, each with its own file dependency graph, evaluated across 7,399 total rubrics",
  "tracking_demand": "Agent must track which of up to 20,476 files across 74 file types are relevant, their dependency relationships, and adaptively retrieve/reason across files to satisfy each task's rubrics.",
  "scoring": "subgoal-checkpoint-partial-credit - evaluated across 7,399 total rubrics (about 19 rubrics per task on average) requiring cross-file retrieval, contextual reasoning, and adaptive decision-making; explicit fine-grained rubric-level partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "The best agent reaches only about 60% on Workspace-Bench, substantially below the human result of 80.7%, with average agent performance at only 45.1%; no specific model is named as best in the abstract.",
  "availability": null,
  "goal_span": "We construct realistic workspaces with 5 worker profiles, 74 file types, 20,476 files (up to 20GB) and curate 388 tasks, each with its own file dependency graph, evaluated across 7,399 total rubrics that require cross-file retrieval, contextual reasoning, and adaptive decision-making.",
  "horizon_span": "We further provide Workspace-Bench-Lite, a 100-task subset that preserves the benchmark distribution while reducing evaluation costs by about 70%.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287962808",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285139845",
  "title": "World of Workflows: a Benchmark for Bringing World Models to Enterprise Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 7,
  "publication_date": "2026-01-29",
  "months_since_pub": 8,
  "citations_per_month": 0.88,
  "artifact_name": "World of Workflows (WoW) / WoW-bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete constrained agentic tasks within a ServiceNow environment governed by 4,000+ business rules and 55 active hidden workflows; predict cascading side effects of actions across interconnected databases; mentally simulate hidden state transitions to avoid silent constraint violations under limited observability",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Actions can trigger cascading side effects across interconnected databases via hidden workflows, so an action satisfying the immediate visible task can silently violate downstream constraints elsewhere in the system.",
  "n_goals": "234 tasks (WoW-bench); 4,000+ business rules and 55 active workflows in the underlying WoW environment",
  "tracking_demand": "Agent must mentally simulate hidden state transitions and predict cascading side effects across interconnected databases to bridge the observability gap, since high-fidelity feedback is often unavailable.",
  "scoring": "other:constrained-task-completion-plus-dynamics-modeling",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce World of Workflows (WoW), a realistic ServiceNow-based environment incorporating 4,000+ business rules and 55 active workflows embedded in the system, alongside WoW-bench, a benchmark of 234 tasks evaluating constrained agentic task completion and enterprise dynamics modeling capabilities.",
  "horizon_span": "alongside WoW-bench, a benchmark of 234 tasks evaluating constrained agentic task completion and enterprise dynamics modeling capabilities.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285139845",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287023600",
  "title": "YC-Bench: Benchmarking AI Agents for Long-Term Planning and Consistent Execution",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 8,
  "publication_date": "2026-04-01",
  "months_since_pub": 5,
  "citations_per_month": 1.6,
  "artifact_name": "YC-Bench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "manage employees within a simulated startup; select task contracts to pursue under uncertainty; maintain profitability against adversarial clients and growing payroll; detect and avoid bankruptcy-inducing failure modes (e.g. adversarial-client mismanagement, over-parallelization) over a one-year run",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Employee management, contract selection, and cash-flow decisions all draw on a shared capital pool that compounds over hundreds of turns, so early mistakes (e.g. misjudging an adversarial client) compound into bankruptcy risk later in the same simulated year.",
  "n_goals": null,
  "tracking_demand": "The agent must persist information across context truncation via a scratchpad (the strongest predictor of success), tracking employees, contracts, cash reserves, and adversarial-client risk across hundreds of turns spanning a simulated year.",
  "scoring": "continuous-reward",
  "horizon_value": "hundreds of turns (over a simulated one-year horizon)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run",
  "horizon_stated": "yes",
  "headline_result": "Claude Opus 4.6 achieves the highest average final funds at $1.27M (starting capital $200K), followed by GLM-5 at $1.21M at 11x lower inference cost; only three of 12 models consistently surpass starting capital.",
  "availability": null,
  "goal_span": "we task an agent with running a simulated startup over a one-year horizon spanning hundreds of turns. The agent must manage employees, select task contracts, and maintain profitability in a partially observable environment where adversarial clients and growing payroll create compounding consequences for poor decisions.",
  "horizon_span": "we introduce YC-Bench, a benchmark that evaluates these capabilities by tasking an agent with running a simulated startup over a one-year horizon spanning hundreds of turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287023600",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "turns",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290831800",
  "title": "What Could the Agent See at 19:05? Generating Temporal Enterprise Scenarios from Real Research and Replaying Them to Evaluate Agents",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2026-08-02",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "temporal enterprise scenario replay system",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "answer questions correctly relative to what data existed and who could see it at a specific queried moment; reason about a persona-driven, temporally-evolving enterprise world spanning many apps; avoid leaking future/hidden record state when reasoning about an earlier moment",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each record's true state at a queried moment depends on the temporal ordering of all prior updates to that record and related records across apps, and correctness requires not leaking any future state hidden inside later record versions.",
  "n_goals": null,
  "tracking_demand": "The agent being evaluated must reason correctly about what data existed and who could see it at a specific queried moment, without conflating it with earlier or later states of the same records across multiple apps.",
  "scoring": "other:snapshot-grounded-correctness-per-moment. No subgoal-checkpoint rubric is described; each replayed moment yields an independently gradable correct answer via a precomputed difference-cache lookup rather than a single whole-episode score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Our system closes two gaps at once: it generates a realistic, persona-driven, temporally-evolving enterprise world from real research, and replays that world at any chosen moment to evaluate any pluggable agent.",
  "horizon_span": "it generates a realistic, persona-driven, temporally-evolving enterprise world from real research, and replays that world at any chosen moment to evaluate any pluggable agent.",
  "extraction_confidence": 1,
  "fit": "reframed",
  "scope_flag": "This is primarily an evaluation-harness/infrastructure paper for temporally consistent grading rather than a benchmark where the agent itself must pursue or track multiple interdependent goals across an episode -- the multi-goal complexity described is in the harness/grading, not necessarily in agent goal-tracking.",
  "url": "https://api.semanticscholar.org/CorpusId:290831800",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285453464",
  "title": "Behavioral Consistency Validation for LLM Agents: An Analysis of Trading-Style Switching through Stock-Market Simulation",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 4,
  "publication_date": "2026-02-02",
  "months_since_pub": 7,
  "citations_per_month": 0.57,
  "artifact_name": "year-long LLM stock-market trading-style simulation testbed",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "trade under a currently assigned style (fundamental or technical) each trading day; periodically (every 10 trading days) reassess and potentially switch trading style based on four behavioral-finance drivers: loss aversion, herding, wealth differentiation, price misalignment; keep style-switching behavior consistent with real-world behavioral-finance theory over the whole simulated year",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The daily trading decisions accumulate wealth and market exposure that feed into the periodic (every-10-day) strategy-reassessment decision, so earlier trading outcomes and stored personality traits constrain which style switches are theoretically justified later.",
  "n_goals": null,
  "tracking_demand": "Agent must process daily price-volume data, retain long-term personality traits (the four behavioral-finance drivers) set at initialization, and track its own accumulated wealth/strategy history to decide whether to switch trading style every 10 days.",
  "scoring": "other:not-stated as task success \u2014 the paper reports alignment metrics (via Mann-Whitney U tests) comparing agents' style-switching behavior to financial theory, i.e. a qualitative-analysis / statistical-comparison scoring rather than binary success or milestone credit.",
  "horizon_value": "year-long (reassessed every 10 trading days)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one year-long simulated trading run, with an internal 10-trading-day reassessment cadence)",
  "horizon_stated": "yes",
  "headline_result": "Recent LLMs' switching behavior is only partially consistent with behavioral-finance theories (no single numeric headline score or human baseline given).",
  "availability": null,
  "goal_span": "In year-long simulations, agents process daily price-volume data, trade under a designated style, and reassess their strategy every 10 trading days.",
  "horizon_span": "In year-long simulations, agents process daily price-volume data, trade under a designated style, and reassess their strategy every 10 trading days.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285453464",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "283883436",
  "title": "AI-Trader: Benchmarking Autonomous Agents in Real-Time Financial Markets",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 19,
  "publication_date": "2025-12-01",
  "months_since_pub": 9,
  "citations_per_month": 2.11,
  "artifact_name": "AI-Trader",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "independently search, verify, and synthesize live market information given only minimal initial context; make live trading decisions (buy/sell/hold) across U.S. stocks, A-shares, and cryptocurrencies at multiple trading granularities; manage risk and sustain positive returns over a continuous live-trading period",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Trading decisions accumulate: an earlier position/risk choice affects the capital and risk exposure available for later live-market decisions, and information gathered in one search/verify cycle feeds into the next trading decision.",
  "n_goals": null,
  "tracking_demand": "Agent must track evolving live market data it has independently searched/verified, current positions/risk exposure across three markets and multiple trading frequencies, and running returns over the continuous live-trading period.",
  "scoring": "continuous-reward - agents are scored on trading returns and risk-management measures over live markets; abstract gives no discrete subgoal-checkpoint partial credit, describing continuous financial performance instead.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "General intelligence does not automatically translate to effective trading capability; most agents exhibit poor returns and weak risk management; risk-control capability determines cross-market robustness; no explicit human/expert baseline is given.",
  "availability": "https://github.com/HKUDS/AI-Trader",
  "goal_span": "Our benchmark implements a revolutionary fully autonomous minimal information paradigm where agents receive only essential context and must independently search, verify, and synthesize live market information without human intervention.",
  "horizon_span": "AI-Trader spans three major financial markets: U.S. stocks, A-shares, and cryptocurrencies, with multiple trading granularities to simulate live financial environments.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283883436",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276580341",
  "title": "Agent Trading Arena: A Study on Numerical Understanding in LLM-Based Agents",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:financial-market-trading",
  "citation_count": 12,
  "publication_date": "2025-02-25",
  "months_since_pub": 19,
  "citations_per_month": 0.63,
  "artifact_name": "Agent Trading Arena",
  "artifact_kind": "environment/simulator",
  "domain": "other:financial-market-trading",
  "goal_types": "make sequential buy/sell trading decisions that directly affect and are affected by shared market prices; compete against other LLM-based agents in a zero-sum stock market; maximize trading performance/returns, especially under high volatility; correctly perform numerical reasoning over price data or chart-based visualizations",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Agents' trading actions directly impact shared market/price dynamics via realistic bid-ask interactions, so one agent's trades change the state (prices) that all other agents subsequently act on -- a zero-sum, mutually influencing resource (capital/shares).",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own portfolio/capital, the current market price state shaped by all agents' recent trades, and historical price patterns, updating its numerical reasoning as the shared market evolves turn to turn.",
  "scoring": "other:trading-performance-comparison-on-NASDAQ-and-CSI-datasets",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Evaluations on NASDAQ and CSI datasets show the proposed method's superiority, particularly under high volatility, with chart-based visualization and a reflection module yielding further improvements; no explicit human-trader baseline is reported.",
  "availability": "https://github.com/wekjsdvnm/Agent-Trading-Arena",
  "goal_span": "we present the Agent Trading Arena, a virtual zero-sum stock market in which LLM-based agents engage in competitive multi-agent trading and directly impact price dynamics.",
  "horizon_span": "Existing approaches are limited to historical backtesting, where trading actions cannot influence market prices and agents train only on static data.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276580341",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279155427",
  "title": "AssetOpsBench: A Real-World Evaluation Benchmark for AI-Driven Task Automation in Industrial Asset Management",
  "year": 2025,
  "venue": "Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 17,
  "publication_date": "2025-06-04",
  "months_since_pub": 15,
  "citations_per_month": 1.13,
  "artifact_name": "AssetOpsBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "orchestrate the correct domain-specific agent(s) (from a catalog of four) to answer a natural-language industrial-operations query; correctly complete condition-monitoring and maintenance-scheduling workflow steps grounded in a simulated IoT environment",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Correctly answering a query typically requires invoking multiple domain-specific agents/tools in the right order and using each tool's observed output to inform the next step, so intermediate steps are coupled by shared, evolving IoT state.",
  "n_goals": "140+ human-authored queries; Plan-Execute agents average approximately 2.6-4.4 steps per task, Agent-As-Tool approximately 4-6+ steps per task",
  "tracking_demand": "The agent must track intermediate tool/agent outputs across a multi-step think-act-observe loop, orchestrate the four domain-specific agents appropriately, and maintain consistency with the simulated CouchDB-backed IoT environment's evolving state.",
  "scoring": "other:architectural-tradeoff-metrics -- three key metrics analyze architectural trade-offs between Agent-As-Tool and Plan-Execute paradigms, plus a systematic failure-mode discovery procedure, rather than a single binary success score; the abstract does not describe fine-grained intra-task subgoal partial credit.",
  "horizon_value": "Plan-Execute agents complete most tasks in approximately 2.6-4.4 steps; Agent-As-Tool agents typically require approximately 4-6+ steps due to its iterative think-act-observe loop",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "AssetOpsBench provides a multimodal ecosystem comprising a catalog of four domain-specific agents, a curated dataset of 140+ human-authored natural-language queries grounded in real industrial scenarios, and a simulated, CouchDB-backed IoT environment. We introduce an automated evaluation framework that uses three key metrics to analyze architectural trade-offs between the Agent-As-Tool and Plan-Execute paradigms, along with a systematic procedure for the automated discovery of emerging failure modes.",
  "horizon_span": "Plan-Execute is consistently more step-efficient: most models complete tasks in \u22482.6-4.4 steps, whereas Agent-As-Tool typically requires \u22484-6+ steps due to its iterative think-act-observe loop.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279155427",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283557102",
  "title": "Automating Complex Document Workflows via Stepwise and Rollback-Enabled Operation Orchestration",
  "year": 2025,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2025-12-04",
  "months_since_pub": 9,
  "citations_per_month": 0.0,
  "artifact_name": "AutoDW",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "execute a sequence of interdependent, user-specified document-editing instructions within a session; keep the execution trajectory aligned with evolving document state and user intent across the whole session; recover from failed API calls/arguments via rollback at both the argument and API level",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Instructions within a session are interdependent (later instructions assume the document state produced by earlier ones), and API actions are conditioned on intent-filtered candidates plus 'the evolving states of the document', with rollback used to undo an action without derailing the rest of the session.",
  "n_goals": "1,708 human-annotated instructions across 250 sessions (~6.8 instructions/session on average)",
  "tracking_demand": "System must track the evolving document state after each API action, the remaining interdependent instructions in the session, and whether any prior action needs to be rolled back (at argument or API level) to stay aligned with user intent.",
  "scoring": "other:mixed \u2014 reports both instruction-level and session-level completion rates (90% and 62% respectively), i.e. explicit partial credit exists at the instruction level within a session even when the whole session is not fully completed.",
  "horizon_value": "1,708 instructions across 250 sessions (~6.8 instructions/session on average)",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per session (episode)",
  "horizon_stated": "yes",
  "headline_result": "AutoDW achieves 90% instruction-level and 62% session-level completion, outperforming strong baselines by 40% and 76% respectively (no human baseline given).",
  "availability": null,
  "goal_span": "AutoDW achieves 90% and 62% completion rates on instruction- and session-level tasks, respectively, outperforming strong baselines by 40% and 76%.",
  "horizon_span": "we construct a comprehensive benchmark of 250 sessions and 1,708 human-annotated instructions, reflecting realistic document processing scenarios with interdependent instructions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283557102",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283737550",
  "title": "CP-Env: Evaluating Large Language Models on Clinical Pathways in a Controllable Hospital Environment",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 6,
  "publication_date": "2025-12-11",
  "months_since_pub": 9,
  "citations_per_month": 0.67,
  "artifact_name": "CP-Env",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "triage a patient correctly to determine care pathway; consult the appropriate specialist given patient information; order and interpret diagnostic tests as needed; participate appropriately in multidisciplinary team meetings; complete a patient's full journey across branching, long-horizon clinical-pathway stages",
  "goal_origin": "mixed:overall-pathway-structure-given-up-front-branching-emitted-by-environment",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Later clinical stages (e.g., diagnostic testing, multidisciplinary team meetings) depend on decisions and information gathered in earlier stages (e.g., triage, specialist consultation), so a hallucinated or lost critical diagnostic detail early in the pathway degrades correctness of later-stage decisions; branching means different pathway routes are possible depending on earlier choices.",
  "n_goals": null,
  "tracking_demand": "The agent must track patient information, diagnostic findings, and evolving care-pathway state across branching stages from triage through specialist consultation, diagnostic testing, and multidisciplinary team meetings.",
  "scoring": "other:three-tiered-evaluation(clinical-efficacy/process-competency/professional-ethics). The paper proposes a three-tiered evaluation framework (Clinical Efficacy, Process Competency, Professional Ethics), an explicit multi-axis, non-binary rubric rather than a single pass/fail outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/SPIRAL-MED/CP_ENV",
  "goal_span": "CP-Env simulates a hospital ecosystem with patient and physician agents, constructing scenarios ranging from triage and specialist consultation to diagnostic testing and multidisciplinary team meetings for agent interaction. Following real hospital adaptive flow of healthcare, it enables branching, long-horizon task execution.",
  "horizon_span": "it enables branching, long-horizon task execution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283737550",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278905020",
  "title": "CRMArena-Pro: Holistic Assessment of LLM Agents Across Diverse Business Scenarios and Interactions",
  "year": 2025,
  "venue": "Trans. Mach. Learn. Res.",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 42,
  "publication_date": "2025-05-24",
  "months_since_pub": 16,
  "citations_per_month": 2.62,
  "artifact_name": "CRMArena-Pro",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of nineteen expert-validated CRM tasks across sales, service, and configure-price-quote (CPQ) processes; sustain correct behavior across multi-turn interactions guided by diverse personas; maintain confidentiality awareness throughout the interaction, for both B2B and B2C scenarios",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Confidentiality awareness and persona-appropriate behavior must be maintained consistently across all turns of a multi-turn interaction, and workflow-execution steps depend on prior turns' disclosures/decisions; single-turn success (58%) substantially exceeds multi-turn success (35%), evidencing later turns are harder to sustain correctly.",
  "n_goals": "nineteen expert-validated tasks across sales, service, and CPQ processes, for both B2B and B2C scenarios",
  "tracking_demand": "The agent must track persona-specific context, confidentiality constraints, and workflow state across multi-turn interactions, since single-turn success (58%) drops substantially to about 35% once genuine multi-turn tracking is required.",
  "scoring": "binary-final-success -- leading LLM agents achieve only around 58% single-turn success, dropping to approximately 35% in multi-turn settings; the abstract separately reports 'Workflow Execution' success (over 83% single-turn) as one business skill among several, implying skill-specific scoring, though not a fine-grained subgoal-checkpoint rubric per se.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Leading LLM agents achieve only ~58% single-turn success, dropping to ~35% in multi-turn settings; Workflow Execution is more tractable (over 83% single-turn) than other business skills; agents show near-zero inherent confidentiality awareness. No human/expert baseline is reported.",
  "availability": null,
  "goal_span": "It distinctively incorporates multi-turn interactions guided by diverse personas and robust confidentiality awareness assessments... Experiments reveal leading LLM agents achieve only around 58% single-turn success on CRMArena-Pro, with performance dropping significantly to approximately 35% in multi-turn settings. While Workflow Execution proves more tractable for top agents (over 83% single-turn success), other evaluated business skills present greater challenges.",
  "horizon_span": "Experiments reveal leading LLM agents achieve only around 58% single-turn success on CRMArena-Pro, with performance dropping significantly to approximately 35% in multi-turn settings.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278905020",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281705844",
  "title": "DRBench: A Realistic Benchmark for Enterprise Deep Research",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 15,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 1.25,
  "artifact_name": "DRBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "identify supporting facts for a multi-step research query from both the public web and a private company knowledge base; synthesize facts drawn from heterogeneous enterprise sources (productivity software, cloud file systems, emails, chat, web) into one answer; produce a coherent, well-structured final report grounded in the retrieved facts",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The final report's coherence and accuracy depend on having gathered and cross-checked facts from multiple heterogeneous sources (web, cloud files, emails, chat) that are individually incomplete, so the synthesis subgoal is constrained by the fact-finding subgoals across all these sources.",
  "n_goals": "100 deep research tasks across 10 domains (e.g. Sales, Cybersecurity, Compliance)",
  "tracking_demand": "The agent must track which facts it has found in which source (public web vs. private enterprise knowledge base), maintain factual accuracy across sources, and assemble a coherent report structure from these accumulated facts.",
  "scoring": "other:recall-accuracy-and-report-quality. Agents are 'evaluated on their ability to recall relevant insights, maintain factual accuracy, and produce coherent, well-structured reports' -- three explicit, separately assessed dimensions, but not framed as a formal subgoal-checkpoint/milestone rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "DRBench evaluates agents on multi-step queries ... that require identifying supporting facts from both the public web and private company knowledge base. Each task is grounded in realistic user personas and enterprise context, spanning a heterogeneous search space that includes productivity software, cloud file systems, emails, chat conversations, and the open web... agents are evaluated on their ability to recall relevant insights, maintain factual accuracy, and produce coherent, well-structured reports.",
  "horizon_span": "We release 100 deep research tasks across 10 domains, such as Sales, Cybersecurity, and Compliance.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281705844",
  "provenance": "web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280635453",
  "title": "DevNous: An LLM-Based Multi-Agent System for Grounding IT Project Management in Unstructured Conversation",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2025-08-12",
  "months_since_pub": 13,
  "citations_per_month": 0.0,
  "artifact_name": "DevNous",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "identify actionable intents from informal, unstructured team-chat dialogue; manage stateful, multi-turn administrative workflows (task formalization, progress-summary synthesis) grounded in that dialogue",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Correctly formalizing a task or synthesizing a progress summary later in the conversation depends on correctly identifying and tracking actionable intents raised earlier in the informal dialogue, since the system must integrate scattered, unstructured references into a single structured artifact.",
  "n_goals": "160 realistic, interactive conversational turns; multi-label ground truth per turn",
  "tracking_demand": "The agent must identify actionable intents across informal chat and maintain stateful workflows (task formalization, progress tracking) across a benchmark of 160 conversational turns, correctly matching a multi-label ground truth per turn.",
  "scoring": "other:exact-match-plus-f1 -- DevNous achieves an exact match turn accuracy of 81.3% and a multiset F1-Score of 0.845, i.e. two distinct metrics (turn-level exact match and multiset F1) rather than one single number, though still per-turn rather than a fine-grained sub-action rubric.",
  "horizon_value": "a new benchmark of 160 realistic, interactive conversational turns",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "dataset-wide total (160 turns across the benchmark, not necessarily one single continuous 160-turn conversation)",
  "horizon_stated": "yes",
  "headline_result": "DevNous achieves an exact match turn accuracy of 81.3% and a multiset F1-Score of 0.845 on the new 160-turn benchmark; no human/expert baseline gap is reported.",
  "availability": null,
  "goal_span": "We introduce DevNous, a Large Language Model-based (LLM) multi-agent expert system, to automate this unstructured-to-structured translation process. DevNous integrates directly into team chat environments, identifying actionable intents from informal dialogue and managing stateful, multi-turn workflows for core administrative tasks like automated task formalization and progress summary synthesis. To quantitatively evaluate the system, we introduce a new benchmark of 160 realistic, interactive conversational turns.",
  "horizon_span": "To quantitatively evaluate the system, we introduce a new benchmark of 160 realistic, interactive conversational turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280635453",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283711860",
  "title": "EcomBench: Towards Holistic Evaluation of Foundation Agents in E-commerce",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 7,
  "publication_date": "2025-12-09",
  "months_since_pub": 9,
  "citations_per_month": 0.78,
  "artifact_name": "EcomBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "retrieve deep, possibly multi-hop information relevant to a real e-commerce user demand; perform multi-step reasoning across e-commerce-domain data; integrate knowledge from multiple sources to resolve a task; correctly complete tasks across three graded difficulty levels",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Higher-difficulty tasks explicitly build on and require lower-level capabilities together: deep information retrieval feeds into multi-step reasoning, which itself must integrate knowledge from multiple sources, so a Level-3 task cannot be solved by treating any one of these capabilities in isolation.",
  "n_goals": "three difficulty levels; task categories drawn from genuine user demands in leading global e-commerce ecosystems",
  "tracking_demand": "The agent must track partial evidence gathered from deep retrieval and multiple knowledge sources across many action steps before it can complete a higher-difficulty task.",
  "scoring": "other:difficulty-tiered-accuracy. The abstract describes three difficulty levels evaluating deep information retrieval, multi-step reasoning, and cross-source knowledge integration, implying tiered scoring by capability level, but does not describe a formal subgoal-checkpoint credit scheme within a single task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "It covers multiple task categories within e-commerce scenarios and defines three difficulty levels that evaluate agents on key capabilities such as deep information retrieval, multi-step reasoning, and cross-source knowledge integration.",
  "horizon_span": "Level 3 tasks cannot be solved in just a few action steps",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283711860",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283896603",
  "title": "Finch: Benchmarking Finance & Accounting across Spreadsheet-Centric Enterprise Workflows",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 9,
  "publication_date": "2025-12-15",
  "months_since_pub": 9,
  "citations_per_month": 1.0,
  "artifact_name": "Finch (FinWorkBench)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "interleave data entry, structuring, formatting, web search, cross-file retrieval, calculation, modeling, validation, translation, visualization, and reporting subtasks into one finance/accounting workflow; correctly complete each of the linked tasks that compose one of 172 composite workflows",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Workflows are composite (172 workflows made of 384 tasks), so tasks within a workflow interleave data entry/retrieval/calculation/modeling/validation/reporting steps that build on each other's outputs (e.g. calculation depends on prior cross-file retrieval), across large numbers of interlinked spreadsheets, PDFs, and other artifacts.",
  "n_goals": "172 composite workflows with 384 tasks, involving 1,710 spreadsheets with 27 million cells",
  "tracking_demand": "Agent must track state across many interlinked spreadsheets, PDFs, and artifacts (27 million spreadsheet cells total) while interleaving retrieval, calculation, modeling, validation, and reporting sub-steps of one composite workflow.",
  "scoring": "other:mixed \u2014 human evaluation reports a workflow-level pass rate (38.4% for GPT-5.1 Pro) alongside per-workflow time spent (16.8 minutes average), i.e. whole-workflow success plus an efficiency metric, without explicit finer-grained per-task checkpoint credit described in the abstract.",
  "horizon_value": "16.8 (GPT-5.1 Pro average)",
  "horizon_unit": "wall-clock-minutes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per workflow (episode)",
  "horizon_stated": "yes",
  "headline_result": "Under human evaluation, GPT-5.1 Pro spends an average of 16.8 minutes per workflow yet passes only 38.4% of workflows (no human-expert baseline time/pass-rate given for comparison).",
  "availability": null,
  "goal_span": "This yields 172 composite workflows with 384 tasks, involving 1,710 spreadsheets with 27 million cells, along with PDFs and other artifacts, capturing the intrinsically messy, long-horizon, knowledge-intensive, and collaborative nature of real-world enterprise work.",
  "horizon_span": "Under human evaluation, GPT-5.1 Pro spends an average of 16.8 minutes per workflow yet passes only 38.4% of workflows.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283896603",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "wall-clock-minutes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277824051",
  "title": "GraphicBench: A Planning Benchmark for Graphic Design with Language Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain-other",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "other:creative-graphic-design-planning",
  "citation_count": 2,
  "publication_date": "2025-04-15",
  "months_since_pub": 17,
  "citations_per_month": 0.12,
  "artifact_name": "GraphicBench",
  "artifact_kind": "benchmark",
  "domain": "other:creative-graphic-design-planning",
  "goal_types": "produce a workflow plan satisfying explicit design constraints stated in a user query; also satisfy implicit commonsense design constraints not stated by the user; select the correct action from 46 available tools at each workflow step; coordinate outputs across three design experts without violating global dependencies",
  "goal_origin": "mixed:explicit-constraints-given-up-front-plus-implicit-commonsense-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Workflow steps must respect spatial relationships and cross-expert global dependencies, so a decision made by one design expert constrains what subsequent steps by other experts can validly do.",
  "n_goals": "1,079 user queries; 46 available actions/tools across three design experts",
  "tracking_demand": "The agent must track explicit and implicit design constraints, the evolving multi-step workflow plan across three design experts, and which of 46 actions remains valid/appropriate at each step.",
  "scoring": "other:workflow-execution-success-with-failure-mode-analysis. Three failure categories are identified (spatial reasoning, cross-expert coordination, action retrieval) but no explicit per-subgoal partial-credit rubric is described; workflows are judged by whether they lead to successful execution outcomes.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce GraphicBench, a new planning benchmark for graphic design that covers 1,079 user queries and input images across four design types. We further present GraphicTown, an LLM agent framework with three design experts and 46 actions (tools) to choose from for executing each step of the planned workflows in web environments.",
  "horizon_span": "GraphicTown, an LLM agent framework with three design experts and 46 actions (tools) to choose from for executing each step of the planned workflows",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277824051",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280000072",
  "title": "HiMA-Ecom: Enabling Joint Training of Hierarchical Multi-Agent E-commerce Assistants",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 8,
  "publication_date": "2025-06-24",
  "months_since_pub": 15,
  "citations_per_month": 0.53,
  "artifact_name": "HiMA-Ecom",
  "artifact_kind": "dataset",
  "domain": "business-office-enterprise",
  "goal_types": "master agent coordinates multiple specialized sub-agents in an e-commerce workflow; each sub-agent pursues a functionally distinct role-specific objective (e.g. recall domain knowledge, execute a service function); jointly optimize system-level behavior via multi-agent reinforcement learning",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The master agent's coordination decisions constrain which sub-agent handles which sub-task, and sub-agents' outputs feed the shared system-level reward, so joint optimization requires reconciling potentially conflicting per-agent objectives.",
  "n_goals": null,
  "tracking_demand": "The master agent must track which specialized sub-agent is responsible for which functional sub-task and how sub-agent outputs compose into overall system behavior, using a collaboratively updated memory across training.",
  "scoring": "other:comparison-against-deepseek-r1-v3-baselines",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "HiMA-R1 built on smaller 3B/7B open-source models achieves performance comparable to DeepSeek-R1 and surpasses DeepSeek-V3 by an average of 6%; no human baseline given.",
  "availability": null,
  "goal_span": "HiMA-Ecom contains 22.8K instances, including agent-specific supervised fine-tuning samples with memory and system-level input-output pairs for joint multi-agent reinforcement learning. ... a master agent coordinates multiple specialized sub-agents",
  "horizon_span": "HiMA-Ecom contains 22.8K instances, including agent-specific supervised fine-tuning samples with memory and system-level input-output pairs for joint multi-agent reinforcement learning.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "This paper's primary contribution is a joint-training method (HiMA-R1/VR-GRPO) rather than a benchmark artifact in its own right; the 'benchmark' here is training/eval data for that method.",
  "url": "https://api.semanticscholar.org/CorpusId:280000072",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281724947",
  "title": "MEMTRACK: Evaluating Long-Term Memory and State Tracking in Multi-Platform Dynamic Agent Environments",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 12,
  "publication_date": "2025-10-01",
  "months_since_pub": 11,
  "citations_per_month": 1.09,
  "artifact_name": "MEMTRACK",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "acquire relevant facts scattered across asynchronous cross-platform events (Slack/Linear/Git); select the correct, currently-valid fact when multiple conflicting/noisy candidates exist; resolve conflicts between contradictory or stale cross-referring information over time",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Facts introduced on one platform (e.g. Slack) may be referenced, updated, or contradicted later on another platform (e.g. Linear, Git), so correctly resolving a query requires linking and reconciling dependent, cross-platform, chronologically-ordered events.",
  "n_goals": null,
  "tracking_demand": "The agent must acquire, select, and reconcile facts from a chronologically platform-interleaved timeline spanning Slack, Linear, and Git, handling noisy, conflicting, and cross-referring information plus codebase/file-system exploration, across long horizons.",
  "scoring": "other:multi-metric -- Correctness, Efficiency, and Redundancy metrics are reported to capture memory-mechanism effectiveness beyond simple QA accuracy; finer-grained than one binary score, though the abstract does not describe true subgoal-checkpoint partial credit within a single instance.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best performing GPT-5 model only achieves a 60% Correctness score on MEMTRACK; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Consequently, our benchmark tests memory capabilities such as acquistion, selection and conflict resolution... We introduce pertinent metrics for Correctness, Efficiency, and Redundancy that capture the effectiveness of memory mechanisms beyond simple QA performance.",
  "horizon_span": "Each benchmark instance provides a chronologically platform-interleaved timeline, with noisy, conflicting, cross-referring information as well as potential codebase/file-system comprehension and exploration.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281724947",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283110353",
  "title": "Mini Amusement Parks (MAPs): A Testbed for Modelling Business Decisions",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2025-11-19",
  "months_since_pub": 10,
  "citations_per_month": 0.2,
  "artifact_name": "Mini Amusement Parks (MAPs)",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "maximize the amusement park's overall value over the given time horizon; make daily operational decisions (building rides/shops, hiring staff, setting a research agenda) that keep the business viable under sparse, stochastic feedback; reason over the park's spatial layout while planning these decisions",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Daily decisions (building, hiring, research) share the same limited resources and spatial layout, and consequences of early decisions compound over the remaining days of the horizon, so agents must actively learn environment dynamics from sparse experience while anticipating long-term, stochastic consequences.",
  "n_goals": "episodes span a 50-day horizon on easy mode, extended to 100 days on medium mode; daily action choices (build, hire, research) recur each day within that horizon",
  "tracking_demand": "The agent must track the park's spatial layout, built rides/shops, staffing, accumulated (sparse) experience about environment dynamics, and overall park value across a 50-100 day episode, planning under uncertainty at each of many daily decision points.",
  "scoring": "continuous-reward -- performance is measured via final park value compared against expert human performance (experts outperform current systems by 11.4x on easy mode and 15.3x on medium mode); the abstract does not describe a subgoal-checkpoint partial-credit rubric beyond this continuous value comparison.",
  "horizon_value": "episodes span a 50-day horizon on easy mode, extended to 100 days on medium mode",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode",
  "horizon_stated": "yes",
  "headline_result": "Experts outperform state-of-the-art agent systems by 11.4x on easy mode and 15.3x on medium mode; persistent weaknesses found in long-horizon planning, sample-efficient learning, spatial reasoning, and modelling uncertainty.",
  "availability": "https://github.com/Skyfall-Research/MAPs",
  "goal_span": "To this end, we introduce Mini Amusement Parks (MAPs), an amusement-park simulator designed to evaluate an agent's ability to model its environment, anticipate long-term consequences under uncertainty, and strategically operate a complex business. We provide expert human performance and a comprehensive evaluation of state-of-the-art agents, finding experts outperform these systems by 11.4x on easy mode and 15.3x on medium mode.",
  "horizon_span": "the horizon is extended from 50 to 100",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283110353",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280635129",
  "title": "OdysseyBench: Evaluating LLM Agents on Long-Horizon Complex Office Application Workflows",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 51,
  "publication_date": "2025-08-12",
  "months_since_pub": 13,
  "citations_per_month": 3.92,
  "artifact_name": "OdysseyBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "identify essential information buried in long-horizon interaction histories; perform multi-step reasoning/actions across Word, Excel, PDF, Email, and Calendar applications; complete real-world-derived tasks (OdysseyBench+) or newly synthesized complex tasks (OdysseyBench-Neo)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later steps depend on information and state accumulated earlier in the long interaction history and must be carried correctly across different office applications.",
  "n_goals": "300 tasks (OdysseyBench+); 302 tasks (OdysseyBench-Neo)",
  "tracking_demand": "Agent must retain and retrieve essential facts from long-horizon interaction histories spanning multiple applications in order to correctly execute later multi-step, cross-application actions.",
  "scoring": "other:not-specified-in-abstract",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Our benchmark comprises two complementary splits: OdysseyBench+ with 300 tasks derived from real-world use cases, and OdysseyBench-Neo with 302 newly synthesized complex tasks. Each task requires agent to identify essential information from long-horizon interaction histories and perform multi-step reasoning across various applications.",
  "horizon_span": "OdysseyBench+ with 300 tasks derived from real-world use cases, and OdysseyBench-Neo with 302 newly synthesized complex tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280635129",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "283458138",
  "title": "PPTArena: A Benchmark for PowerPoint Editing",
  "year": 2025,
  "venue": "",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 7,
  "publication_date": "2025-12-02",
  "months_since_pub": 9,
  "citations_per_month": 0.78,
  "artifact_name": "PPTArena",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "apply each of a deck's human-curated edits correctly from natural-language instructions; maintain layout-sensitive and cross-slide consistency across the whole deck while editing; plan and verify edit sequences via an iterative plan-edit-check loop (for the PPTPilot agent)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Edits within a deck are often 'compound, layout-sensitive, and cross-slide', so an edit applied to one slide/element can affect layout or consistency elsewhere in the same deck, requiring the agent to verify each result before moving to the next edit.",
  "n_goals": "over 1,300 human-curated edits across 100 decks (~13 edits/deck on average), spanning 2,125 slides",
  "tracking_demand": "Agent must track the deck's current structural/visual state after each edit, verify each edit against a ground-truth rubric, and maintain deck-wide consistency (styles, cross-slide references) across the full sequence of edits.",
  "scoring": "milestone-rubric \u2014 'Each edit pairs a ground-truth deck with a target rubric and is scored by two Vision-Language Model (VLM) judges', giving explicit per-edit (subgoal-level) rubric-based credit.",
  "horizon_value": "~13 edits per deck (1,300+ edits across 100 decks)",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per deck (episode)",
  "horizon_stated": "yes",
  "headline_result": "PPTPilot outperforms strong VLM-based agents by more than 10 percentage points on compound, layout-sensitive, and cross-slide edits, though 'all agents still struggle on long-horizon, document-scale tasks' (no human baseline given).",
  "availability": "https://github.com/michaelofengend/PPTArena",
  "goal_span": "PPTArena features 100 decks with over 1,300 human-curated edits across 2,125 slides, spanning text, charts, animations, and professional master styles.",
  "horizon_span": "PPTArena features 100 decks with over 1,300 human-curated edits across 2,125 slides",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283458138",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284513100",
  "title": "ProSoftArena: Benchmarking Hierarchical Capabilities of Multimodal Agents in Professional Software Environments",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 5,
  "publication_date": "2025-12-30",
  "months_since_pub": 9,
  "citations_per_month": 0.56,
  "artifact_name": "ProSoftArena",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "operate professional software tools according to a hierarchical capability level (L1-L3); complete realistic work/research tasks spanning 6 disciplines and 13 core professional applications; coordinate across multiple professional software applications for L3 multi-software workflows",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The benchmark establishes 'the first capability hierarchy tailored to agent use of professional software,' where L3 tasks require multi-software workflows built on top of L1/L2 single-application capabilities, so failing at a lower capability level blocks completion of the higher-level, cross-application L3 tasks.",
  "n_goals": "436 realistic work and research tasks spanning 6 disciplines and 13 core professional applications, organized into a capability hierarchy (L1-L3)",
  "tracking_demand": "The agent must track its progress within and across professional-software applications as required by the task's capability level, since L3 tasks require coordinating state across multiple applications rather than operating one application in isolation.",
  "scoring": "other:hierarchical-capability-level-success-rate. The best-performing agent attains only 24.4% success on L2 tasks and 'completely fails' on L3 multi-software workflow, i.e. the paper reports success separately by capability level (L1/L2/L3), a tiered (subgoal-level) credit rather than one flat score.",
  "horizon_value": "human execution steps grow from an average of 5.1 steps (14.8 seconds) at L1 to 86.9 steps (506.8 seconds) at L3",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task (average, broken down by capability level L1 vs L3)",
  "horizon_stated": "yes",
  "headline_result": "The best-performing agent attains only a 24.4% success rate on L2 tasks and completely fails on L3 multi-software workflows; no human/expert baseline given in the abstract.",
  "availability": "https://prosoftarena.github.io",
  "goal_span": "We establish the first capability hierarchy tailored to agent use of professional software and construct a benchmark of 436 realistic work and research tasks spanning 6 disciplines and 13 core professional applications... Extensive experiments show that even the best-performing agent attains only a 24.4% success rate on L2 tasks and completely fails on L3 multi-software workflow.",
  "horizon_span": "human execution steps and time... from an average of 5.1 steps and 14.8 seconds at L1 to 86.9 steps and 506.8 seconds at L3",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284513100",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280527371",
  "title": "REALM-Bench: A Benchmark for Evaluating Multi-Agent Systems on Real-world, Dynamic Planning and Scheduling Tasks",
  "year": 2025,
  "venue": "Knowledge Discovery and Data Mining",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 12,
  "publication_date": "2025-02-26",
  "months_since_pub": 19,
  "citations_per_month": 0.63,
  "artifact_name": "REALM-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "solve each of 14 real-world planning/scheduling problems with multiple parallel planning threads; maintain feasibility across inter-agent dependencies as the problem scales in complexity; adapt schedules in real time when unexpected disruptions arrive",
  "goal_origin": "mixed:given-up-front-problems-plus-injected-mid-episode-disruptions",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Parallel planning threads share inter-agent dependencies, so a disruption or decision in one thread can require replanning across dependent threads.",
  "n_goals": "14 planning and scheduling problems",
  "tracking_demand": "Agents must track the state of parallel planning threads, inter-dependencies between agents/threads, and disruption events in order to replan in real time.",
  "scoring": "other:multi-metric-comparison-against-15-baselines - scored via 2+ evaluation metrics against 15 baseline planning methods; abstract does not describe explicit subgoal-level checkpoint credit beyond per-problem metrics.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "The suite encompasses 14 designed planning and scheduling problems that progress from basic to highly complex, incorporating key aspects such as multi-agent coordination, inter-agent dependencies, and dynamic environmental disruptions.",
  "horizon_span": "Each problem can be scaled along three dimensions: the number of parallel planning threads, the complexity of inter-dependencies, and the frequency of unexpected disruptions requiring real-time adaptation.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280527371",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282592087",
  "title": "Remote Labor Index: Measuring AI Automation of Remote Work",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 21,
  "publication_date": "2025-10-30",
  "months_since_pub": 11,
  "citations_per_month": 1.91,
  "artifact_name": "Remote Labor Index (RLI)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete a whole real, economically valuable freelance/remote-work project end to end; achieve a level of automation comparable to what a human freelancer would deliver for that project",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each project is a 'real-world, economically valuable' whole engagement rather than an atomic task, implying it bundles multiple dependent deliverables/sub-steps whose combined completion determines the reported automation rate, though the abstract does not spell out the internal sub-task dependency structure.",
  "n_goals": null,
  "tracking_demand": "Agent must track progress toward completing an entire multi-part real-world project (not a single isolated action) to be credited with automating that unit of remote labor.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports an 'automation rate' (2.5% for the best agent), which functions like a graded/partial measure of how much of the real-world labor value was captured, though the exact grading mechanism per project is not detailed.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The highest-performing agent achieves an automation rate of only 2.5%; AI agents perform near the floor on RLI overall (no specific model named as best in the abstract).",
  "availability": null,
  "goal_span": "we introduce the Remote Labor Index (RLI), a broadly multi-sector benchmark comprising real-world, economically valuable projects designed to evaluate end-to-end agent performance in practical settings.",
  "horizon_span": "AIs have made rapid progress on research-oriented benchmarks of knowledge and reasoning, but it remains unclear how these gains translate into economic value and automation.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282592087",
  "provenance": "forward-citation,web-registry",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "281682720",
  "title": "SCUBA: Salesforce Computer Use Benchmark",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "business-office-enterprise",
  "citation_count": 7,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 0.58,
  "artifact_name": "SCUBA",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "navigate a specific enterprise software UI (Salesforce) to accomplish a CRM task; manipulate data records and automate workflows within the Salesforce platform; retrieve information and troubleshoot issues as part of a realistic CRM task; generalize across three personas (platform administrators, sales representatives, service agents) each with distinct task profiles",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Milestone progress in SCUBA is tracked step-by-step through a task's UI navigation, data manipulation, and workflow-automation steps, so completing a later milestone typically depends on the state changes produced by earlier steps within the same Salesforce sandbox session.",
  "n_goals": "300 task instances derived from real user interviews, spanning 3 personas",
  "tracking_demand": "The agent must track its progress toward fine-grained milestones within each Salesforce sandbox task (UI navigation state, data changes made, workflow steps completed) to receive interpretable milestone-progress credit.",
  "scoring": "subgoal-checkpoint-partial-credit. SCUBA operates with 'fine-grained evaluation metrics to capture milestone progress,' explicit partial credit for completing intermediate milestones within a task, not just a final binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "In the zero-shot setting, open-source-model computer-use agents that perform well on OSWorld achieve less than 5% success on SCUBA, while closed-source-model-based methods reach up to 39%; demonstration-augmented settings raise success to 50% while reducing time/cost by 13%/16%. No human/expert baseline given.",
  "availability": null,
  "goal_span": "SCUBA operates in Salesforce sandbox environments with support for parallel execution and fine-grained evaluation metrics to capture milestone progress. We benchmark a diverse set of agents under both zero-shot and demonstration-augmented settings.",
  "horizon_span": "SCUBA contains 300 task instances derived from real user interviews, spanning three primary personas, platform administrators, sales representatives, and service agents.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281682720",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279260580",
  "title": "SOP-Bench: Complex Industrial SOPs for Evaluating LLM Agents",
  "year": 2025,
  "venue": "Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 20,
  "publication_date": "2025-06-09",
  "months_since_pub": 15,
  "citations_per_month": 1.33,
  "artifact_name": "SOP-Bench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "correctly execute each step of a complex, multi-step Standard Operating Procedure; orchestrate the correct tools/APIs at each SOP step; produce ground-truth outputs matching the SOP's authored specification across 12 business domains",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "SOP steps are procedurally ordered by construction (a Standard Operating Procedure), and the paper reports agents invoke incorrect tools nearly 100% of the time when the tool registry is much larger than needed, indicating that tool-selection errors at one step propagate into failure of the overall procedure.",
  "n_goals": "2,000+ tasks across 12 business domains",
  "tracking_demand": "The agent must track its position within a multi-step SOP, select the correct tool from a registry that may contain many irrelevant tools, and maintain consistency with the SOP's ground-truth interface and outputs across the whole procedure.",
  "scoring": "other:average-task-success-rate-by-agent-type",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Function-Calling and ReAct agents achieve average task success rates of only 27% and 48% respectively (per escalated search), with the Claude 4 family outperforming Claude 4.5 on ReAct tasks in some conditions per the abstract's illustrative example; no human baseline given.",
  "availability": "https://github.com/amazon-science/sop-bench",
  "goal_span": "LLM-based agents struggle to execute complex, multi-step Standard Operating Procedures (SOPs) that are fundamental to industrial automation. ... We introduce SOP-Bench, a benchmark of 2,000+ tasks from human expert-authored SOPs across 12 business domains ... yielding realistic tasks with executable interfaces and ground-truth outputs.",
  "horizon_span": "We introduce SOP-Bench, a benchmark of 2,000+ tasks from human expert-authored SOPs across 12 business domains",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279260580",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281724665",
  "title": "StockBench: Can LLM Agents Trade Stocks Profitably In Real-world Markets?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 33,
  "publication_date": "2025-10-02",
  "months_since_pub": 11,
  "citations_per_month": 3.0,
  "artifact_name": "STOCKBENCH",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "make a sequential daily buy/sell/hold decision based on incoming market signals (prices, fundamentals, news); maximize cumulative return across the whole multi-month trading period; manage risk, minimizing maximum drawdown and maintaining a strong Sortino ratio",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each daily buy/sell/hold decision changes the portfolio position that the next day's decision is made from, so earlier decisions constrain later ones through accumulated position, cash, and risk exposure across the multi-month trading environment.",
  "n_goals": null,
  "tracking_demand": "Agent must track its current portfolio position, cumulative return, and risk exposure (drawdown) as it processes a new daily market signal (prices, fundamentals, news) each step of a multi-month simulated trading run.",
  "scoring": "continuous-reward \u2014 'Performance is measured using financial metrics such as cumulative return, maximum drawdown, and the Sortino ratio', a continuous multi-metric measure rather than a single binary success/failure or discrete subgoal checkpoints.",
  "horizon_value": "multi-month (exact day/month count not given)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per episode (one multi-month trading run, with daily decision granularity)",
  "horizon_stated": "yes",
  "headline_result": "Most models struggle to outperform the simple buy-and-hold baseline, though some demonstrate potential for higher returns and stronger risk management (no specific numeric score or human baseline given).",
  "availability": "open-sourced (per abstract: 'We release STOCKBENCH as an open-source benchmark'); no specific URL given.",
  "goal_span": "Agents receive daily market signals -- including prices, fundamentals, and news -- and make sequential buy, sell, or hold decisions.",
  "horizon_span": "STOCKBENCH, a contamination-free benchmark designed to evaluate LLM agents in realistic, multi-month stock trading environments. Agents receive daily market signals... and make sequential buy, sell, or hold decisions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281724665",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280337218",
  "title": "StaffPro: an LLM Agent for Joint Staffing and Profiling",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 2,
  "publication_date": "2025-07-29",
  "months_since_pub": 14,
  "citations_per_month": 0.14,
  "artifact_name": "StaffPro",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "assign and schedule tasks to workers (staffing), forming teams as needed; continuously estimate workers' latent skills, preferences, and other attributes from unstructured feedback (profiling); optimize staffing performance over time as profiling estimates improve via an ongoing human-agent feedback loop",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Staffing decisions (who is assigned to which task/team) depend on the current profile estimate of each worker's latent attributes, while profiling estimates are themselves updated based on feedback resulting from prior staffing decisions, creating a continuous feedback loop between the two coupled goals.",
  "n_goals": null,
  "tracking_demand": "StaffPro must track each worker's evolving latent-attribute profile (estimated from ongoing human feedback) and the current staffing/schedule state, updating both jointly over the 'life-long' profiling horizon.",
  "scoring": "other:consulting-firm-simulation-case-study - demonstrated via a consulting-firm simulation example showing StaffPro 'successfully estimates workers' attributes and generates high quality schedules'; abstract does not describe a formal quantitative benchmark score or subgoal-checkpoint credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "By analyzing human feedback, our agent continuously estimates the latent features of workers, realizing life-long worker profiling and ensuring optimal staffing performance over time.",
  "horizon_span": "realizing life-long worker profiling and ensuring optimal staffing performance over time.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "StaffPro is demonstrated via a single consulting-firm simulation example rather than a graded benchmark with a defined scoring protocol across many cases; it may be closer to a system/framework paper than a benchmark artifact.",
  "url": "https://api.semanticscholar.org/CorpusId:280337218",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283072266",
  "title": "UpBench: A Dynamically Evolving Real-World Labor-Market Agentic Benchmark Framework Built for Human-Centric AI",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 4,
  "publication_date": "2025-11-15",
  "months_since_pub": 10,
  "citations_per_month": 0.4,
  "artifact_name": "UpBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete a real, verified client transaction job sourced from the Upwork labor marketplace; satisfy each of a job's detailed, expert-decomposed, verifiable acceptance criteria; follow instructions faithfully enough to earn positive fine-grained, per-criterion human expert feedback",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each job is decomposed into detailed acceptance criteria assessed individually, but a real client transaction's overall success depends on jointly satisfying enough of these criteria to be judged as meeting genuine professional standards for that job.",
  "n_goals": "one job per task, each decomposed into detailed, verifiable acceptance criteria (exact per-job criterion count not given); the task set is regularly refreshed",
  "tracking_demand": "Agent must track which of a job's detailed acceptance criteria it has satisfied, informed by expert freelancer decomposition, to produce a submission gradeable on fine-grained, per-criterion feedback rather than a binary pass/fail.",
  "scoring": "subgoal-checkpoint-partial-credit - a rubric-based evaluation framework in which expert freelancers decompose each job into detailed, verifiable acceptance criteria and assess AI submissions with per-criterion feedback, enabling fine-grained analysis beyond binary pass/fail; an explicit per-criterion partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "UpBench employs a rubric-based evaluation framework, in which expert freelancers decompose each job into detailed, verifiable acceptance criteria and assess AI submissions with per-criterion feedback. This structure enables fine-grained analysis of model strengths, weaknesses, and instruction-following fidelity beyond binary pass/fail metrics.",
  "horizon_span": "Each task corresponds to a verified client transaction, anchoring evaluation in genuine work activity and financial outcomes.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283072266",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276575302",
  "title": "Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 67,
  "publication_date": "2025-02-20",
  "months_since_pub": 19,
  "citations_per_month": 3.53,
  "artifact_name": "Vending-Bench",
  "artifact_kind": "environment/simulator",
  "domain": "business-office-enterprise",
  "goal_types": "balance inventory levels of vending-machine stock; place restocking orders from suppliers; set item prices; pay recurring daily fees; sustain a profitable long-running vending-machine business without derailing",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Inventory levels, cash on hand, and pending orders form a shared resource pool that constrains pricing and ordering decisions, so mismanaging one (e.g., misreading a delivery schedule) cascades into the others.",
  "n_goals": null,
  "tracking_demand": "The agent must continuously track inventory counts, cash/profit, outstanding orders and their delivery schedules, and daily fees across a run spanning over 20M tokens, since forgetting an order or misreading a schedule causes derailment.",
  "scoring": "continuous-reward",
  "horizon_value": ">20M",
  "horizon_unit": "other:tokens-per-run",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run (a single continuous vending-machine management run)",
  "horizon_stated": "yes",
  "headline_result": "Claude 3.5 Sonnet and o3-mini manage the machine well in most runs and turn a profit; no explicit human/expert profit baseline is reported in the abstract.",
  "availability": null,
  "goal_span": "Agents must balance inventories, place orders, set prices, and handle daily fees - tasks that are each simple but collectively, over long horizons (>20M tokens per run) stress an LLM's capacity for sustained, coherent decision-making.",
  "horizon_span": "over long horizons (>20M tokens per run) stress an LLM's capacity for sustained, coherent decision-making",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276575302",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281682820",
  "title": "VitaBench: Benchmarking LLM Agents with Versatile Interactive Tasks in Real-world Applications",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 35,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 2.92,
  "artifact_name": "VitaBench",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "satisfy each of multiple real user requests combined into one cross-scenario task (food delivery, in-store consumption, online travel); reason across temporal and spatial dimensions while using a large tool set (66 tools); proactively clarify ambiguous instructions and track shifting user intent across a multi-turn conversation",
  "goal_origin": "mixed:given-up-front-with-user-intent-shifting-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "Each cross-scenario task is 'derived from multiple real user requests', which the agent must jointly satisfy using a shared pool of 66 tools spanning multiple domain-specific policies, so tool/resource use for one request can constrain what remains available or consistent for another request in the same task.",
  "n_goals": "100 cross-scenario tasks (main results) and 300 single-scenario tasks, each built from multiple real user requests, using 66 tools",
  "tracking_demand": "Agent must track shifting user intent across a multi-turn conversation, temporal and spatial constraints, and the state of a large (66-tool) toolset spanning multiple life-serving domains simultaneously.",
  "scoring": "other:not-stated precisely \u2014 the abstract introduces 'a rubric-based sliding window evaluator, enabling robust assessment of diverse solution pathways', which allows for graded assessment of different valid trajectories rather than a single golden path, though explicit subgoal-checkpoint credit is not spelled out.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the most advanced models achieve only 30% success rate on cross-scenario tasks, and less than 50% success rate on others (no human baseline given).",
  "availability": "https://vitabench.github.io/",
  "goal_span": "Each task is derived from multiple real user requests and requires agents to reason across temporal and spatial dimensions, utilize complex tool sets, proactively clarify ambiguous instructions, and track shifting user intent throughout multi-turn conversations.",
  "horizon_span": "yielding 100 cross-scenario tasks (main results) and 300 single-scenario tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281682820",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "283458556",
  "title": "Benchmarking LLM Agents for Wealth-Management Workflows",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2025-12-01",
  "months_since_pub": 9,
  "citations_per_month": 0.0,
  "artifact_name": "Wealth-Management benchmark (extension of TheAgentCompany)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "complete each of 12 wealth-management task-pairs spanning retrieval, analysis, and synthesis/communication; satisfy explicit acceptance criteria under deterministic graders; operate correctly under both high- and low-autonomy task variants",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each task-pair chains retrieval, analysis, and synthesis/communication sub-steps, so errors or omissions in the retrieval stage propagate into the analysis and communication stages that depend on it.",
  "n_goals": "12 task-pairs spanning retrieval, analysis, and synthesis/communication",
  "tracking_demand": "Agent must track retrieved data, intermediate analysis results, and explicit acceptance criteria across the retrieval-analysis-synthesis pipeline, plus which autonomy variant (high vs. low) governs how independently it may act.",
  "scoring": "other:deterministic-graders-with-acceptance-criteria - each task-pair has explicit acceptance criteria assessed by deterministic graders, implying checkable per-criterion credit rather than a single opaque LLM-judged score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We construct a benchmark of 12 task-pairs for wealth management assistants spanning retrieval, analysis, and synthesis/communication, with explicit acceptance criteria and deterministic graders.",
  "horizon_span": "This study introduces synthetic domain data, enriches colleague simulations, and prototypes an automatic task-generation pipeline.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283458556",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "LLM-judge or expert rubric",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284078900",
  "title": "Does It Tie Out? Towards Autonomous Legal Agents in Venture Capital",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "business-office-enterprise",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "business-office-enterprise",
  "citation_count": 0,
  "publication_date": "2025-12-21",
  "months_since_pub": 9,
  "citations_per_month": 0.0,
  "artifact_name": "capitalization tie-out benchmark (legal AI)",
  "artifact_kind": "benchmark",
  "domain": "business-office-enterprise",
  "goal_types": "verify that every security (shares, options, warrants) is supported by underlying legal documentation; verify that every issuance term (vesting schedules, acceleration triggers, transfer restrictions) is consistent across the dataroom; maintain strict evidence traceability while reconciling thousands of pages of legal documents; produce a deterministic, fully-reconciled capitalization table",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "All securities and issuance terms must be reconciled against the same underlying dataroom of legal documents, so verifying one security may require cross-referencing evidence also used for others, and the overall capitalization table is only valid if every individual verification is simultaneously consistent.",
  "n_goals": "184 to 1,292 individual securities requiring verification across company stages (Seed to Series B); dataroom sizes from hundreds to tens of thousands of pages",
  "tracking_demand": "The agent must track which securities and issuance terms have been verified against which supporting documents, maintaining strict evidence traceability across a growing dataroom (up to tens of thousands of pages) as it reconciles a scaling number of individual securities.",
  "scoring": "other:deterministic-reconciliation-accuracy. The task requires 'deterministic outputs,' i.e. the reconciled cap table must exactly match ground truth; the paper does not describe a subgoal-checkpoint partial-credit rubric distinct from this exact-match requirement, though the analysis of per-security/per-step workload implies fine-grained tracking.",
  "horizon_value": "workload grows from approximately 2,700 atomic verification steps at Seed stage to nearly 8,000 steps at Series B stage",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per company's tie-out reconciliation, scaling with company stage/complexity)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "verifying that every security (for example, shares, options, warrants) and issuance term (for example, vesting schedules, acceleration triggers, transfer restrictions) is supported by large sets of underlying legal documentation. While LLMs continue to improve on legal benchmarks, specialized legal workflows, such as capitalization tie-out, remain out of reach even for strong agentic systems. The task requires multi-document reasoning, strict evidence traceability, and deterministic outputs.",
  "horizon_span": "Fig. 7 quantifies this burden by tracking the total number of atomic 'steps' executed by the counsel to complete the tie-out... We observe a near-tripling of workload, from approximately 2,700 steps at Seed to nearly 8,000 steps at Series B.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284078900",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291470039",
  "title": "AcCoRD: Evaluating User-Agent Collaboration Under Realistic User Preference Dynamics",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-08-28",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "AcCoRD",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "resolve underspecified user preferences in online shopping or travel planning; detect and satisfy preferences that emerge mid-interaction (not stated upfront); adapt to preferences the user later adjusts or relaxes during the interaction",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Because preferences 'are formed, revealed, adjusted, and relaxed during interaction', later turns can invalidate or refine constraints established earlier, so the agent must continuously reconcile new preference information against everything gathered previously in the same session.",
  "n_goals": null,
  "tracking_demand": "Agent must maintain and continuously update a model of the user's preferences as they are formed, revealed, adjusted, or relaxed turn-by-turn, and recognize when uncertainty about a preference needs to be resolved.",
  "scoring": "other:not-stated \u2014 the abstract reports that models 'struggle to satisfy preferences that emerge or evolve mid-interaction' but does not describe a specific partial-credit/checkpoint scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier models can handle underspecification but struggle to satisfy preferences that emerge or evolve mid-interaction; prompting alone fails to elicit the required uncertainty recognition (no specific numeric score or human baseline given).",
  "availability": null,
  "goal_span": "We introduce AcCoRD, a user-agent collaboration benchmark requiring agents to handle diverse user preference dynamics in two domains: online shopping and travel planning.",
  "horizon_span": "We evaluate five frontier LLMs under two prompting strategies: vanilla ReAct, and an uncertainty-guided variant",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291470039",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "291334588",
  "title": "AgentWorld: Personality-Aware Reliability Evaluation for Agentic Information Retrieval",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-08-25",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "AgentWorld",
  "artifact_kind": "environment/simulator",
  "domain": "customer-service-dialogue",
  "goal_types": "maintain consistent, reliable tool-use behavior across interactions with users of varying personality (Big Five/OCEAN) profiles; handle dual-control handoffs correctly between agent and (simulated) user/other controller; resist or survive adversarial perturbations to a required-intermediate-state 'spine' of the task without brittle failure; achieve consistent pass rates (pass^k) across repeated attempts rather than succeeding only once by chance",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The Risk Analyser 'snapshots required-intermediate-state spines' and branches Monte-Carlo rollouts under perturbations, meaning task success depends on a specific sequence of intermediate states being reached correctly, and perturbing an early intermediate state can cascade into failure at later spine states.",
  "n_goals": "3 experiments: a conversational analytics agent across 10 OCEAN personas (240 evaluator judgments); a customer-support agent across 5 tasks x 4 persona variants; adversarial stress-testing of 5 tasks",
  "tracking_demand": "The agent must track its stateful tool-use progress consistently across repeated attempts and varying user personas, while the Risk Analyser separately tracks the required-intermediate-state spine of a task to evaluate how perturbations at each state affect eventual outcomes.",
  "scoring": "other:pass-at-k-with-partial-credit-and-risk-scoring. The framework combines 'the pass^k consistency metric with structured fault classification, partial-credit scoring, and dual-control handoff verification' plus a separate risk score (Dempster-Shafer fusion, Shapley attribution) -- partial credit and process-level risk analysis, not a single binary success measure.",
  "horizon_value": "each persona ran 3 multi-turn conversation exchanges against the analytics agent, producing 60 messages total (30 from the agent, 30 from personas), across 10 personas",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per persona (per one persona's evaluation run in the conversational-analytics-agent experiment)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "Three experiments demonstrate the framework: a conversational analytics agent across 10 OCEAN personas (240 evaluator judgments); a customer-support agent across 5 tasks x 4 persona variants; and adversarial stress-testing of 5 tasks revealing pre-existing trajectory brittleness ($V_{\\min}=0.375$ without perturbation) and tool/infrastructure-layer attack dominance (Shapley: 46% system, 38% action).",
  "horizon_span": "each persona ran 3 multi-turn conversation exchanges against the analytics agent, producing 60 messages (30 from the agent, 30 from personas)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291334588",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288861296",
  "title": "CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-06-01",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "CRAB-Bench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "reason over a constraint graph spanning multiple interdependent entities to find a solution among thousands of misleading candidate distractors; engage in multi-turn dialogue with a realistic (non-cooperative, persona-driven) simulated user rather than a cooperative template-like one; accommodate multiple valid solutions rather than a single fixed correct answer",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Task constraints form a graph over multiple interdependent entities, so satisfying one constraint often narrows or eliminates candidate solutions relevant to satisfying the others, and structured distractors mean an early wrong inference can lead the agent down an unrecoverable path among thousands of misleading candidates.",
  "n_goals": null,
  "tracking_demand": "Agent must track which constraints (graph edges/entities) have been satisfied so far, which candidate solutions remain viable given thousands of distractors, and information gathered/disclosed across a multi-turn dialogue with a realistic user persona.",
  "scoring": "other:pass-at-1-with-behavioral-dimension-breakdown - reports pass@1 (best model 61% under cooperative simulation, dropping up to 57% further under RUSE) plus behavioral-dimension analysis (e.g., Information Disclosure); a multi-dimensional scoring scheme beyond simple task success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "The best of four frontier LLM agents achieves only 61% pass@1 on CRAB-Bench under cooperative user simulation, and switching to the more realistic RUSE user simulator causes further drops of up to 57%, concentrated in task-solving ability; no human/expert baseline is given.",
  "availability": null,
  "goal_span": "CRAB-Bench generates tasks via a constraint graph over multiple interdependent entities with structured distractors, requiring agents to reason carefully over thousands of misleading candidates where only a tiny fraction of solutions are valid.",
  "horizon_span": "CRAB-Bench generates tasks via a constraint graph over multiple interdependent entities with structured distractors, requiring agents to reason carefully over thousands of misleading candidates where only a tiny fraction of solutions are valid.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288861296",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290623013",
  "title": "CallBench: A Benchmark for Dual-Goal Coordination in Phone Call Assistants",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 1,
  "publication_date": "2026-06-22",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "CallBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "satisfy the device owner's explicit preset goal for the call; correctly infer and respond to the caller's implicit and dynamic goal; make turn-level decisions that correctly reconcile these two goals under alignment, complementarity, irrelevance, or conflict relations; adhere to preset instructions while maintaining dialogue quality, safety, and rhythm",
  "goal_origin": "mixed:owner-preset-goal-given-plus-caller-goal-emitted-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The owner's preset goal and the caller's implicit, dynamic goal can align, complement, be irrelevant to, or conflict with each other, and this relation can itself evolve turn by turn, so the assistant's response at one turn must be reconciled against both goals' accumulated state rather than either goal in isolation.",
  "n_goals": "50,000 multi-turn phone call dialogues across 6 scenarios (takeout, delivery, taxi, work, life, harassment); regular/emergent/no-preset cases",
  "tracking_demand": "The assistant must track the owner's preset goal, the evolving implicit goal of the caller, and the current relation between them (alignment/complementarity/irrelevance/conflict) turn by turn across the dialogue, to make reliable turn-level decisions between the two goals.",
  "scoring": "other:preset-aware-turn-level-multi-dimensional-evaluation. The benchmark uses a 'preset-aware turn-level evaluation protocol covering semantic understanding, context use, active guidance, response quality, preset compliance, dialogue rhythm, and safety,' multi-dimensional, turn-level scoring rather than a single milestone/subgoal-checkpoint rubric.",
  "horizon_value": "average of 5.322 turns per dialogue across all 50,000 dialogues (scenario averages ranging from 4.437 to 7.241 turns)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per dialogue)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce \\textsc{CallBench}, a Chinese benchmark for evaluating dual-goal coordination in phone call assistants. \\textsc{CallBench} contains 50,000 complete multi-turn phone call dialogues across six scenarios... It covers regular presets, emergent presets, and no-preset cases, and includes diverse relations between owner-side and caller-side goals, such as alignment, complementarity, irrelevance, and conflict.",
  "horizon_span": "We report the number of dialogues and the average turns per dialogue.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290623013",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291217808",
  "title": "FraudBench: Stress-Testing Policy-Grounded Banking Agents Against Adaptive Fraud",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-08-02",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "FraudBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "safely act on caller requests during a banking conversation while checking authorization, identity, and policy compliance at every step; retrieve applicable rules from a 698-document internal policy corpus before permitting a sensitive action; detect and refuse chained/adaptive fraud attempts where an earlier probe or admission makes a later, superficially valid request unsafe",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Safety is explicitly history-dependent: single-control tasks satisfy every precondition but one, and adaptive attacks make a later, locally-valid request unsafe because of an earlier probe, admission, or failed attempt, so the agent must remember the full conversation state, not just the current turn.",
  "n_goals": "150 authored adversarial scenarios (107 in the frozen graded set: 90 across ten fraud mechanisms + 17 chained adaptive attacks; 43 further held out)",
  "tracking_demand": "The agent must track the caller's identity/authorization claims, any tool access it has already granted, and prior probes/admissions across the conversation, checking each new request against the accumulated history and a 698-document policy corpus.",
  "scoring": "other:attack-security-rate. A preliminary single-trial evaluation of four agents on the 107 graded tasks yields attack-security between 49% and 65%; the abstract does not describe subgoal-level partial credit distinct from this per-scenario security outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Attack-security between 49% and 65% across four evaluated agents on the 107 graded tasks, with money-mule and first-party fraud the most common cross-model weaknesses; no human/expert baseline given.",
  "availability": null,
  "goal_span": "Safety is history-dependent: single-control tasks satisfy every precondition but one, and adaptive attacks make a later, locally valid request unsafe because of an earlier probe, admission, or failed attempt. Each scenario is annotated with observable evidence, prohibited actions, safe dispositions, and intervention points.",
  "horizon_span": "FraudBench contains 150 authored adversarial scenarios; a frozen public set of 107 (90 across ten fraud mechanisms plus 17 chained adaptive attacks) is used for all reported runs, with 43 further chained attacks held out.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291217808",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289624384",
  "title": "IHBench: Evaluating Post-Interruption Recovery in Voice Agents with Structured Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 4,
  "publication_date": "2026-06-17",
  "months_since_pub": 3,
  "citations_per_month": 1.33,
  "artifact_name": "IHBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "resume a state-machine-driven workflow at the correct step after a user interruption; address the content of the user's interjection; avoid re-delivering content the user already heard; achieve task fulfillment across 10 enterprise domains despite one of 6 injected interruption types",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "An interruption forces the agent to reconcile the workflow's prior state-machine step with a new interjection: it must resume the correct step, respond to the interjection, and avoid repeating already-delivered content, so these demands are jointly evaluated rather than independent.",
  "n_goals": "6 interruption types; 10 enterprise domains; 27 evaluated audio-language-model configurations",
  "tracking_demand": "The agent must track its position in a state-machine-driven workflow, which content has already been delivered to the user, and the content of the user's interjection, in order to both recover to the correct step and adequately address the interruption.",
  "scoring": "other:two-axis-per-interruption-rubric-task-fulfillment-and-recovery-quality",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Closed-weight models are consistently more robust to interruptions than open-weight ones, winning far more often on task fulfillment and degrading roughly 3.3x more slowly as conversations grow longer; no explicit human-expert ceiling score is given (a human study instead validates the LLM judge).",
  "availability": null,
  "goal_span": "Each interruption is scored on two axes: task fulfillment and recovery quality.",
  "horizon_span": "they win far more often on task fulfillment, degrade roughly 3.3x more slowly as conversations grow longer, and show no audio-versus-text modality gap",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289624384",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284475748",
  "title": "Beyond IVR: Benchmarking Customer Support LLM Agents for Business-Adherence",
  "year": 2026,
  "venue": "Conference of the European Chapter of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 11,
  "publication_date": "2026-01-02",
  "months_since_pub": 8,
  "citations_per_month": 1.38,
  "artifact_name": "JourneyBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "adhere to multi-step business policies/SOPs throughout a support conversation; navigate task dependencies correctly as the conversation unfolds; remain robust to unpredictable user/environment behavior while covering the required user-journey graph paths",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Policy-driven steps in the support conversation have dependencies (later steps require earlier ones to be correctly resolved per the SOP graph), and deviating from policy order reduces measured User Journey Coverage.",
  "n_goals": "703 conversations across three domains; graph-based scenario structure (exact node/path count not stated in the abstract)",
  "tracking_demand": "The agent must track which policy-graph nodes/steps it has satisfied so far, adhere to multi-step business rules, and adapt to unpredictable user/environment behavior across the conversation.",
  "scoring": "milestone-rubric -- the User Journey Coverage Score explicitly measures policy adherence as graph-path coverage, i.e. partial, path-level credit rather than a single binary success/failure per conversation.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "DPA significantly boosts policy adherence, even allowing smaller models like GPT-4o-mini to outperform larger ones like GPT-4o; no human-expert baseline or specific numeric gap is reported.",
  "availability": null,
  "goal_span": "JourneyBench leverages graph representations to generate diverse, realistic support scenarios and proposes the User Journey Coverage Score, a novel metric to measure policy adherence.",
  "horizon_span": "Across 703 conversations in three domains, we show that DPA significantly boosts policy adherence, even allowing smaller models like GPT-4o-mini to outperform more capable ones like GPT-4o.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284475748",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290466694",
  "title": "LLMs Get Lost in Evolving User Intent",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-07-22",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": null,
  "artifact_kind": "scenario-suite",
  "domain": "customer-service-dialogue",
  "goal_types": "track a single user goal that is incrementally revealed across conversation turns; detect and adopt goal revisions as the user changes their mind mid-conversation; detect and follow mid-conversation redirections of the original goal; re-run an existing single-turn benchmark's task under this evolving-intent protocol without new annotation",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Each new turn can reveal, revise, or redirect the same underlying goal, so an agent that commits to an earlier, incomplete version of the goal must overwrite/update its plan rather than pursue two versions in parallel.",
  "n_goals": "up to 7 turns per conversation, including the initial turn (transition types repeated twice each)",
  "tracking_demand": "The agent must maintain a running model of a single, evolving user intent across turns, discarding or revising earlier partial specifications as more of the goal is revealed, revised, or redirected.",
  "scoring": "other:reuses-each-source-benchmarks-original-evaluation-protocol",
  "horizon_value": "up to 7 (including the initial turn)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per conversation/task",
  "horizon_stated": "yes",
  "headline_result": "Strong static-setting performance does not transfer to the evolving-intent setting, with substantial performance drops across model families; no single figure or human baseline given.",
  "availability": "https://github.com/microsoft/evolving-intent",
  "goal_span": "we introduce a framework that transforms static, single-turn tasks into dynamic multi-turn conversations in which the user's intent evolves across turns--incrementally revealed, revised, and at times redirected mid-conversation--while preserving each task's original evaluation protocol",
  "horizon_span": "We combine these transition types such that each occurs twice within a conversation, resulting in up to 7 turns, including the initial turn.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": "This paper's object of study is a single evolving goal (not multiple concurrent subgoals), and it is a meta-framework applied to other benchmarks' tasks rather than a standalone artifact with its own fixed domain -- may not fit a multi-subgoal corpus criterion as well as purpose-built multi-goal benchmarks.",
  "url": "https://api.semanticscholar.org/CorpusId:290466694",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290758586",
  "title": "Research on a Multi-SOP Interruption\u2013Resumption Agentic Algorithm for Complex Business Workflows",
  "year": 2026,
  "venue": "Journal of Artificial Intelligence Practice",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": null,
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": null,
  "artifact_kind": "environment/simulator",
  "domain": "customer-service-dialogue",
  "goal_types": "complete a customer-requested Standard Operating Procedure (SOP) workflow; switch to and complete a different SOP mid-interaction when the customer changes topic; resume a previously interrupted SOP without restarting or re-asking for already-given information; detect and recover from stale/rolled-back state via diff reasoning against frozen snapshots",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A six-tuple Global State Container is the single source of truth for cross-SOP information transfer, so information gathered while executing one SOP must be correctly reused when the customer switches to or resumes another SOP, and a Diff Reasoning mechanism must detect state discrepancies between frozen snapshots and current memory to choose lossless recovery vs. safe rollback.",
  "n_goals": null,
  "tracking_demand": "The system must track which SOP is currently active, a shared six-tuple Global State Container of cross-SOP business entities, and frozen-snapshot vs. current-state diffs to support interruption/resumption without re-asking the user for information already given.",
  "scoring": "other:operational-efficiency-and-recovery-metrics. The paper reports an 85% reduction in repeated user inputs (to only 12.3 on average), a 96.3% cross-SOP recovery success rate, and a 73.5% successful chained-task completion rate -- multiple distinct operational metrics rather than one binary score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "The multi-SOP framework reduces repeated user input by 85% (to only 12.3 on average), achieves a 96.3% cross-SOP recovery success rate, and a 73.5% successful chained-task completion rate; no human/expert baseline given.",
  "availability": null,
  "goal_span": "The multi-SOP framework reduced the need for users to input the same information more than once, resulting in an 85% reduction to only 12.3, a cross-SOP recovery success rate of 96.3%, and a 73.5% successful chained task completion rate.",
  "horizon_span": "Average end -to-end time",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290758586",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287352178",
  "title": "SAGE: A Service Agent Graph-guided Evaluation Benchmark",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 3,
  "publication_date": "2026-04-10",
  "months_since_pub": 5,
  "citations_per_month": 0.6,
  "artifact_name": "SAGE (Service Agent Graph-guided Evaluation) / SAGE-Bench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "follow every step of an unstructured Standard Operating Procedure formalized as a Dynamic Dialogue Graph; correctly classify diverse/adversarial user intents and derive the correct subsequent action for each",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Path coverage through the Dynamic Dialogue Graph requires each SOP step to be reached in the order dictated by prior classified intents; an incorrect early intent classification derails downstream required actions (the paper's 'Execution Gap').",
  "n_goals": null,
  "tracking_demand": "The agent must track which SOP-graph node/path it is currently on, the user's evolving (possibly adversarial) intent, and maintain logical compliance across the full dialogue, evaluated at increasing dialogue depths (turns 1, 5, 10, 15, and final turn).",
  "scoring": "milestone-rubric -- a Judge Agent and Rule Engine generate deterministic ground truth verifying logical compliance and path coverage through the SOP graph at multiple checkpoints, rather than a single end-of-dialogue binary score.",
  "horizon_value": "dialogue depths evaluated at turns 1, 5, 10, 15, and the final turn, to measure stability over extended interactions",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/dialogue (turn-depth checkpoints within one conversation)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://anonymous.4open.science/r/SAGE-Bench-4CD3/",
  "goal_span": "SAGE formalizes unstructured SOPs into Dynamic Dialogue Graphs, enabling precise verification of logical compliance and comprehensive path coverage... Evaluation is conducted via a framework where Judge Agents and a Rule Engine analyze interactions between User and Service Agents to generate deterministic ground truth.",
  "horizon_span": "To evaluate model stability over extended interactions, we analyze performance variations across different dialogue depths (Turn 1, 5, 10, 15).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287352178",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289685078",
  "title": "SEATauBench: Adapting Tool-Agent-User Evaluation Into Low-Resource Southeast Asian Languages",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 1,
  "publication_date": null,
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "SEATauBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "carry out tau2-Bench-style tool-agent-user tasks correctly when the conversation language changes; carry out the same tasks correctly as tool specifications are localized into the target SEA language; carry out the same tasks correctly as the task domain itself is localized",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "The three localization settings (conversation language, tool-spec language, domain localization) are progressively stacked, so degradation compounds as more of the interaction is localized simultaneously ('quality and robustness degrade sharply as more task contexts are localized').",
  "n_goals": "5 languages (Mandarin, Vietnamese, Thai, Indonesian, Filipino) x progressively localized settings",
  "tracking_demand": "Agent must correctly interpret and act on user requests, tool specifications, and domain content in the target language while adhering to the policy constraints originally defined in tau2-Bench.",
  "scoring": "other:not-stated \u2014 the abstract reports comparative degradation across settings and languages but does not describe a partial-credit/checkpoint scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "github.com/SEACrowd/SEATauBench",
  "goal_span": "SEATauBench adapts \u03c4 2 -Bench to five languages\u2014Mandarin, Vietnamese, Thai, In-donesian, and Filipino\u2014and evaluates agents across progressively localized settings that vary the language of user-agent interaction, tool specifications, and task domains",
  "horizon_span": "SEATauBench provides a diagnostic benchmark and reusable adaptation pipeline for building reliable multilingual agents for linguistically diverse regions",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289685078",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "291383085",
  "title": "SpeechGym: An Audio-Native Gym for Training Voice Agents via Reinforcement Learning",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-08-26",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "SpeechGym",
  "artifact_kind": "environment/simulator",
  "domain": "customer-service-dialogue",
  "goal_types": "correctly hear and call tools/fill argument slots based on values perceived from native audio (no ASR/TTS); hold multi-turn dialogue entirely through speech to complete the same tasks as an established text agentic benchmark; avoid unauthorized write actions under an insistent caller's social pressure",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "A single misheard argument value cascades into a failed tool call, a retry of the same call, and a wasted step budget, so perceptual accuracy at one turn directly constrains success and efficiency at subsequent turns.",
  "n_goals": null,
  "tracking_demand": "Agent must track dialogue state and correctly perceived argument values purely from native audio across a multi-turn tool-use session, since a single misheard value causes cascading failures that consume the step budget.",
  "scoring": "other:per-turn-process-reward-plus-task-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Trained with per-turn process reward, the agent transfers with no further tuning to an independently implemented voice benchmark, more than doubling task success and carrying an open-weights model from last place to second on that leaderboard, using fewer turns and tokens than before training; no human baseline stated.",
  "availability": null,
  "goal_span": "that single error cascades into a failed call, a retry of the same call, and a wasted step budget... Outcome-only GRPO is gradient-starved here, since almost every rollout group fails identically, while a per-turn process reward crediting each successful tool call restores variance to nearly every group.",
  "horizon_span": "that single error cascades into a failed call, a retry of the same call, and a wasted step budget... using fewer turns and tokens than before training",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291383085",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289131864",
  "title": "T1-Bench: Benchmarking Multi-Scenario Agents in Real-World Domains",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 0,
  "publication_date": "2026-06-09",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "T1-Bench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "complete interleaved customer-facing scenarios spanning 25 domains within one interaction; conduct structured reasoning across multi-turn user-assistant interactions; correctly use tools while maintaining conversational quality across compositionally complex, interwoven scenario threads",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Interleaved scenarios require the agent to track and correctly interweave reasoning threads across concurrently active domains within a single multi-turn interaction, so progress in one domain thread must be preserved while attending to another.",
  "n_goals": "25 domains of varying difficulty",
  "tracking_demand": "Agent must track tool use, conversational state, and multiple concurrent reasoning threads across interleaved scenarios spanning 25 domains within multi-turn user-assistant interactions.",
  "scoring": "other:automatic-evaluation-plus-human-judgment",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "featuring interleaved scenarios that require structured reasoning across multi-turn user-assistant interactions and substantially increasing both compositional complexity and evaluative rigor across 25 domains of varying difficulty... We further complement automatic evaluation with human judgments to strengthen the assessment of qualitative performance.",
  "horizon_span": "featuring interleaved scenarios that require structured reasoning across multi-turn user-assistant interactions and substantially increasing both compositional complexity and evaluative rigor across 25 domains of varying difficulty.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289131864",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285973556",
  "title": "Assessing Risks of Large Language Models in Mental Health Support: A Framework for Automated Clinical AI Red Teaming",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 3,
  "publication_date": "2026-02-23",
  "months_since_pub": 7,
  "citations_per_month": 0.43,
  "artifact_name": "clinical AI red-teaming framework (unnamed)",
  "artifact_kind": "environment/simulator",
  "domain": "customer-service-dialogue",
  "goal_types": "conduct a full therapy session with a simulated patient agent having a dynamic cognitive-affective model; maintain quality of care across the session per a comprehensive risk ontology; de-escalate suicide risk appropriately when it arises during the session; avoid validating patient delusions ('AI Psychosis')",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "The simulated patient's cognitive-affective state evolves based on the AI psychotherapist's prior responses, so an earlier response that validates a delusion or fails to de-escalate risk changes the patient's subsequent state and requires appropriate follow-up in later turns of the same session.",
  "n_goals": "N=369 simulated sessions; 15 patient personas across diverse clinical phenotypes; 6 AI agents evaluated",
  "tracking_demand": "The AI psychotherapist agent must track the patient's evolving cognitive-affective state, emerging risk signals (e.g., suicide risk, delusion reinforcement), and quality-of-care obligations across the whole therapy session.",
  "scoring": "other:quality-of-care-and-risk-ontology-audit. Sessions are assessed against a comprehensive quality-of-care and risk ontology via a stakeholder-validated dashboard, rather than a single binary success score; specific iatrogenic risks (e.g., delusion validation, failed suicide de-escalation) are separately identified.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We apply this framework to a high-impact test case, Alcohol Use Disorder, evaluating six AI agents (including ChatGPT, Gemini, and Character AI) against a clinically-validated cohort of 15 patient personas representing diverse clinical phenotypes. Our large-scale simulation (N=369 sessions) reveals critical safety gaps in the use of AI for mental health support.",
  "horizon_span": "current safety benchmarks often fail to detect the complex, longitudinal risks inherent in therapeutic dialogue",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285973556",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286568451",
  "title": "\u03c4-Voice: Benchmarking Full-Duplex Voice Agents on Real-World Domains",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 19,
  "publication_date": "2026-03-14",
  "months_since_pub": 6,
  "citations_per_month": 3.17,
  "artifact_name": "tau-Voice",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "complete a verifiable grounded task (extended from tau2-bench) via complex multi-turn conversation; adhere to domain policies throughout the conversation; correctly interact with/act on the environment (not just converse) to satisfy the task",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The agent must maintain policy adherence and environment-state consistency across the full multi-turn conversation; earlier turns' actions and disclosures constrain what remains valid or necessary in later turns.",
  "n_goals": "278 tasks total, evaluated under both full-duplex voice and half-duplex text conditions",
  "tracking_demand": "The agent must track conversational state, domain-policy constraints, and environment actions across complex multi-turn conversations, while also managing real-time full-duplex audio interaction (turn-taking, interruptions, accents, noise) alongside the underlying task goals.",
  "scoring": "binary-final-success -- task completion is measured as pass@1 on verifiable grounded tasks, alongside separate voice-interaction-quality metrics; the abstract does not describe additional subgoal-checkpoint partial credit beyond per-task completion.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "GPT-5 (reasoning) achieves 85% task completion (pass@1) in text; voice agents reach only 31-51% under clean conditions and 26-38% under realistic noisy/accented conditions, retaining only 30-45% of text capability. No human/expert baseline is reported.",
  "availability": null,
  "goal_span": "We introduce $\\tau$-voice, a benchmark for evaluating voice agents on grounded tasks with real-world complexity: agents must navigate complex multi-turn conversations, adhere to domain policies, and interact with the environment.",
  "horizon_span": "We evaluate task completion (pass@1) and voice interaction quality across 278 tasks: while GPT-5 (reasoning) achieves 85%, voice agents reach only 31--51% under clean conditions and 26--38% under realistic conditions with noise and diverse accents.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286568451",
  "provenance": "forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284704232",
  "title": "User-Oriented Multi-Turn Dialogue Generation with Tool Use at scale",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 2,
  "publication_date": "2026-01-13",
  "months_since_pub": 8,
  "citations_per_month": 0.25,
  "artifact_name": "unnamed user-oriented multi-turn tool-use dialogue generation pipeline",
  "artifact_kind": "dataset",
  "domain": "customer-service-dialogue",
  "goal_types": "complete multiple distinct task completions accumulating within a single long conversational trajectory; respond appropriately to a simulated user's incremental, turn-by-turn requests and feedback rather than resolving the whole task in one shot",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Because multiple task completions are facilitated within a single trajectory, later tasks/turns build on the conversational and tool-use state accumulated from earlier turns, and the user simulator's incremental, feedback-driven requests mean the agent must track what has already been resolved versus what remains open.",
  "n_goals": "a data-generation pipeline producing 'high turn-count conversations' with multiple task completions per trajectory; exact per-conversation turn/task counts not stated in the abstract",
  "tracking_demand": "The agent must track the state of multiple, potentially overlapping task completions within one long dialogue, correctly interpreting a user simulator's incremental, turn-by-turn requests and feedback rather than resolving everything from an initial fully-specified prompt.",
  "scoring": "other:not-an-evaluation-benchmark -- this paper is a data-generation pipeline producing training data, not itself an evaluation benchmark with a stated scoring rubric; no subgoal-checkpoint or binary success criterion is described in the abstract.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "To bridge this gap, we shift toward a user-oriented simulation paradigm. By decoupling task generation from a dedicated user simulator that mimics human behavioral rules - such as incremental request-making and turn-by-turn feedback - we facilitate more authentic, extended multi-turn dialogues that reflect the iterative nature of real-world problem solving... by facilitating multiple task completions within a single trajectory, it yields a high-density dataset that reflects the multifaceted demands of real-world human-agent interaction.",
  "horizon_span": "we facilitate more authentic, extended multi-turn dialogues that reflect the iterative nature of real-world problem solving... it yields a high-density dataset that reflects the multifaceted demands of real-world human-agent interaction.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": "This is a data-generation pipeline/dataset for training, not itself a benchmark/environment that evaluates agents against goals; may be out of scope for a corpus focused on evaluable artifacts.",
  "url": "https://api.semanticscholar.org/CorpusId:284704232",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "286239096",
  "title": "\u03c4-Knowledge: Evaluating Conversational Agents over Unstructured Knowledge",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 25,
  "publication_date": "2026-03-04",
  "months_since_pub": 6,
  "citations_per_month": 4.17,
  "artifact_name": "\u03c4-Knowledge (\u03c4-Banking domain)",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "retrieve the correct policy document(s) from a densely interlinked, ~700-document knowledge base; coordinate retrieved natural-language knowledge with tool outputs to execute a policy-compliant account update; produce a verifiable, policy-compliant state change during a live customer-support interaction",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Correct execution requires first retrieving the right documents among ~700 'interconnected knowledge documents' and then applying that knowledge correctly to a tool-mediated account update, so a wrong or incomplete retrieval step directly causes an incorrect downstream state change.",
  "n_goals": null,
  "tracking_demand": "Agent must track which of many interconnected banking policy documents are relevant to the current request, reconcile them with live tool outputs, and ensure the resulting account update remains policy-compliant and verifiable.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports 'pass^1' accuracy (~25.5%) with reliability 'degrading sharply over repeated trials', which functions like a binary-final-success metric per trial rather than explicit subgoal partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even frontier models with high reasoning budgets achieve only ~25.5% pass^1, with reliability degrading sharply over repeated trials (no human baseline given).",
  "availability": null,
  "goal_span": "Our new domain, $\\tau$-Banking, models realistic fintech customer support workflows in which agents must navigate roughly 700 interconnected knowledge documents while executing tool-mediated account updates.",
  "horizon_span": "agents must navigate roughly 700 interconnected knowledge documents while executing tool-mediated account updates",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286239096",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "282246655",
  "title": "AgentChangeBench: A Multi-Dimensional Evaluation Framework for Goal-Shift Robustness in Conversational AI",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 3,
  "publication_date": "2025-10-20",
  "months_since_pub": 11,
  "citations_per_month": 0.27,
  "artifact_name": "AgentChangeBench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "complete the original task objective before a mid-dialogue goal shift; recognize and adapt to a mid-dialogue goal shift triggered by one of five user personas; recover task success after a goal shift within enterprise domains (e.g., airline booking, retail); use tools efficiently and non-redundantly while adapting to shifting goals",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "A goal shift injected mid-dialogue supersedes or modifies the original objective, so the agent must abandon or revise prior task state and tool-call plans without discarding progress that remains valid, and continued tool use after the shift is judged for redundancy against the new goal.",
  "n_goals": "2,835 task sequences; five user personas; three enterprise domains",
  "tracking_demand": "The agent must track the original task objective, detect when a mid-dialogue goal shift has occurred, measure its own recovery latency (Goal-Shift Recovery Time), and avoid redundant tool calls (Tool Call Redundancy Rate) while re-establishing progress toward the new goal.",
  "scoring": "other:four-metric-framework-TSR-TUE-TCRR-GSRT",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "GPT-4o reaches 92.2% Goal-Shift Recovery on airline booking shifts while Gemini collapses to 48.6%; retail tasks show near-perfect parameter validity yet redundancy rates above 80%; no human/expert baseline is reported (comparison is model-vs-model).",
  "availability": null,
  "goal_span": "Our framework formalizes evaluation through four complementary metrics: Task Success Rate (TSR) for effectiveness, Tool Use Efficiency (TUE) for reliability, Tool Call Redundancy Rate (TCRR) for wasted effort, and Goal-Shift Recovery Time (GSRT) for adaptation latency.",
  "horizon_span": "AgentChangeBench comprises 2,835 task sequences and five user personas, each designed to trigger realistic shift points in ongoing workflows",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282246655",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "280151307",
  "title": "ECom-Bench: Can LLM Agent Resolve Real-World E-commerce Customer Support Issues?",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 30,
  "publication_date": "2025-07-08",
  "months_since_pub": 14,
  "citations_per_month": 2.14,
  "artifact_name": "ECom-Bench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "resolve a real-world e-commerce customer-support issue via multimodal interaction; adapt to a dynamic, persona-driven simulated user across the dialogue; handle diverse business scenarios reflecting real-world complexity; achieve consistent success across repeated trials (pass^3 metric)",
  "goal_origin": "mixed:issue-given-up-front-user-persona-dynamics-emitted-by-environment",
  "decomposition": "sequential-chain",
  "interdependence": "The dynamic user simulation responds to the agent's prior turns based on persona information, so resolving the underlying issue requires correctly tracking evolving user needs/state across the dialogue, and repeated trials (pass^3) test whether success is consistent rather than a lucky single run.",
  "n_goals": null,
  "tracking_demand": "The agent must track the evolving state of a persona-driven simulated customer's issue, multimodal evidence presented during the conversation, and consistency of resolution across repeated trials.",
  "scoring": "other:pass^k-metric(pass3). The benchmark explicitly uses a pass^3 metric (consistency across 3 trials), e.g., GPT-4o achieves only 10-20% pass^3, a stricter, repeatability-aware metric distinct from simple one-shot binary success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Even advanced models like GPT-4o achieve only a 10-20% pass^3 metric on ECom-Bench; no human/expert baseline given.",
  "availability": "https://github.com/XiaoduoAILab/ECom-Bench",
  "goal_span": "ECom-Bench features dynamic user simulation based on persona information collected from real e-commerce customer interactions and a realistic task dataset derived from authentic e-commerce dialogues. These tasks, covering a wide range of business scenarios, are designed to reflect real-world complexities... even advanced models like GPT-4o achieve only a 10-20% pass^3 metric in our benchmark.",
  "horizon_span": "even advanced models like GPT-4o achieve only a 10-20% pass^3 metric in our benchmark",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280151307",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282246329",
  "title": "Food4All: An Agentic Framework and Benchmark for Food Resource Navigation with Adaptive User Understanding",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 1,
  "publication_date": "2025-10-21",
  "months_since_pub": 11,
  "citations_per_month": 0.09,
  "artifact_name": "Food4All",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "ground a user's underspecified/noisy help-seeking dialogue into a valid resource recommendation; retrieve resources satisfying single food needs; satisfy composite cases with access or document constraints; handle non-ideal user interaction traits (unreasonable demands, rambling, impatience, incomplete answers, inconsistent information) while completing the referral",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Correct referral depends on first correctly grounding requirements and retrieving valid resources; composite cases add access/document constraints that must jointly hold, so an error at grounding or retrieval propagates into referral correctness.",
  "n_goals": "300 multi-turn evaluation tasks",
  "tracking_demand": "The agent must track grounded requirements (schedule, eligibility, intake, document constraints), the set of valid retrieved resources it must preserve into the final recommendation, and the user's non-ideal interaction trait across the dialogue.",
  "scoring": "other:multi-dimensional-metrics-requirement-grounding-retrieval-referral-correctness-efficiency",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The strongest model achieves 96.33% referral accuracy, though diagnostics reveal persistent failures in grounding schedule/eligibility/intake/document constraints; no human/expert baseline is given.",
  "availability": null,
  "goal_span": "We evaluate six Large Language Models (LLMs) on requirement grounding, resource retrieval, final referral correctness, and interaction efficiency.",
  "horizon_span": "couples a food-specific search tool with 300 multi-turn evaluation tasks spanning single food needs, composite cases with access or document constraints, and five non-ideal user interaction traits",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282246329",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "275757481",
  "title": "IntellAgent: A Multi-Agent Framework for Evaluating Conversational AI Systems",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 26,
  "publication_date": "2025-01-19",
  "months_since_pub": 20,
  "citations_per_month": 1.3,
  "artifact_name": "IntellAgent",
  "artifact_kind": "scenario-suite",
  "domain": "customer-service-dialogue",
  "goal_types": "navigate a multi-turn dialogue while integrating domain-specific APIs; adhere to strict, graph-modeled policy constraints throughout the conversation; handle realistic, policy-driven event scenarios generated at varying complexity levels",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "IntellAgent employs a graph-based policy model representing relationships, likelihoods, and complexities of policy interactions, so satisfying one policy constraint in a multi-turn dialogue can interact with (enable or conflict with) other policy constraints represented as graph edges, and event generation composes these into realistic multi-policy scenarios.",
  "n_goals": null,
  "tracking_demand": "The agent must track which of several interacting domain-specific policy constraints apply at each point in a multi-turn dialogue, per a graph-based policy model, while integrating API calls consistent with those constraints.",
  "scoring": "other:fine-grained-diagnostics-via-policy-driven-graph-modeling",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/plurai-ai/intellagent",
  "goal_span": "IntellAgent automates the creation of diverse, synthetic benchmarks by combining policy-driven graph modeling, realistic event generation, and interactive user-agent simulations. ... it employs a graph-based policy model to represent relationships, likelihoods, and complexities of policy interactions, enabling highly detailed diagnostics.",
  "horizon_span": "Conversational AI systems, which must navigate multi-turn dialogues, integrate domain-specific APIs, and adhere to strict policy constraints.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:275757481",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279306202",
  "title": "Effective Red-Teaming of Policy-Adherent Agents",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 11,
  "publication_date": "2025-06-11",
  "months_since_pub": 15,
  "citations_per_month": 0.73,
  "artifact_name": "tau-break",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "adhere consistently to domain policies (e.g. refund eligibility, cancellation rules) across a customer-service conversation; correctly refuse any request that would violate policy; remain helpful and natural while resisting persuasive, policy-aware adversarial pressure from CRAFT",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each new adversarial turn tests whether the policy-adherent agent maintains consistency with all prior policy commitments in the same conversation; a single concession that violates policy at any turn constitutes failure, so earlier turns constrain what remains admissible later.",
  "n_goals": null,
  "tracking_demand": "The agent must track which policies apply to the current request and remain consistent with them across up to 30 dialogue turns, resisting cumulative persuasive pressure (emotional manipulation, coercive framing) from an adversarial user.",
  "scoring": "binary-final-success",
  "horizon_value": "up to 30 dialogue turns",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per conversation/episode",
  "horizon_stated": "yes",
  "headline_result": "CRAFT outperforms conventional jailbreak methods (DAN prompts, emotional manipulation, coercion) at undermining policy-adherent agents; defense strategies provide only partial protection; no single numeric success rate given in the abstract.",
  "availability": "https://github.com/IBM/CRAFT",
  "goal_span": "we present CRAFT, a multi-agent red-teaming system that leverages policy-aware persuasive strategies to undermine a policy-adherent agent in a customer-service scenario ... we introduce tau-break, a complementary benchmark designed to rigorously assess the agent's robustness against manipulative user behavior.",
  "horizon_span": "Each agent interaction spans up to 30 dialogue turns, with seed set to 10 for reproducibility.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279306202",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279251284",
  "title": "\u03c42-Bench: Evaluating Conversational Agents in a Dual-Control Environment",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "customer-service-dialogue",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "customer-service-dialogue",
  "citation_count": 456,
  "publication_date": "2025-06-09",
  "months_since_pub": 15,
  "citations_per_month": 30.4,
  "artifact_name": "\u03c4\u00b2-Bench",
  "artifact_kind": "benchmark",
  "domain": "customer-service-dialogue",
  "goal_types": "coordinate actions with an active user who also uses tools to modify a shared, dynamic environment (dual-control); complete diverse, compositionally-generated verifiable tasks built from atomic components; guide/communicate with the user effectively, not only reason internally; correctly attribute/avoid errors arising from reasoning vs communication/coordination failures",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Because both agent and user can act on the shared world (modeled as a Dec-POMDP), the agent's plan must account for actions the user might independently take, so task state is jointly and interdependently modified by two acting parties rather than the agent alone.",
  "n_goals": null,
  "tracking_demand": "The agent must track the shared dynamic world state as modified by both itself and the user, communicate effectively to guide the user's tool use, and maintain a compositional task's atomic sub-requirements across the dual-control interaction.",
  "scoring": "other:fine-grained-ablation-separating-reasoning-vs-communication-coordination-errors",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Agents show significant performance drops when shifting from no-user (single-control) to dual-control settings; no specific numeric top score or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "Fine-grained analysis of agent performance through multiple ablations including separating errors arising from reasoning vs communication/coordination... our experiments show significant performance drops when agents shift from no-user to dual-control, highlighting the challenges of guiding users.",
  "horizon_span": "A compositional task generator that programmatically creates diverse, verifiable tasks from atomic components, ensuring domain coverage and controlled complexity",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279251284",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288253883",
  "title": "TeachArena: Are Language Agents Ready for Realistic Teaching Work?",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "education-tutoring",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:education-teaching-workflows",
  "citation_count": 0,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "TeachArena",
  "artifact_kind": "benchmark",
  "domain": "other:education-teaching-workflows",
  "goal_types": "infer a warranted pedagogical/teaching decision from evidence (professional pedagogical judgment); adapt tutoring support as the learner's state changes across a multi-turn tutoring session (situated multi-turn tutoring); carry an instructor's request through a learning-management system (LMS) to a completed, verified intervention (end-to-end LMS teaching workflow)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "End-to-end LMS teaching-workflow execution depends on the situated tutoring decisions made along the way, which in turn depend on sound pedagogical judgment about the learner's evolving state, so the three evaluated surfaces (judgment, tutoring, workflow) are layered and later surfaces depend on getting earlier ones right.",
  "n_goals": "354 audited tasks; three complementary evaluated surfaces (pedagogical judgment, situated tutoring, end-to-end LMS workflows)",
  "tracking_demand": "The agent must track the learner's evolving state across a multi-turn tutoring session, evidence supporting a pedagogical insight, and the instructor's request as it is carried through to a persistent, verified artifact or environment state in the LMS.",
  "scoring": "other:matched-verifiers-over-turn-level-responses-tutoring-trajectories-and-persistent-artifacts",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier models show generally capable bounded pedagogical judgment but fall short of professional teaching standards in situated tutoring and end-to-end teaching-workflow execution; no single specific numeric score or human-expert baseline percentage is given in the abstract.",
  "availability": null,
  "goal_span": "Its 354 audited tasks are each built around a pedagogical insight, grounded in evidence, and evaluated with matched verifiers over observable turn-level responses, tutoring trajectories, and persistent artifacts or environment states.",
  "horizon_span": "they must infer a warranted teaching decision from evidence, adapt support as learner state changes, and carry an instructor's request through a learning-management system (LMS) to a completed, verified intervention",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288253883",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287916043",
  "title": "DeepTutor: Towards Agentic Personalized Tutoring",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "education-tutoring",
  "family_source": "extracted-domain-other",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": true,
  "domain_detail": "other:education-tutoring",
  "citation_count": 0,
  "publication_date": "2026-04-10",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "TutorBench",
  "artifact_kind": "benchmark",
  "domain": "other:education-tutoring",
  "goal_types": "deliver citation-grounded tutoring on a specific problem; generate difficulty-calibrated follow-up questions matched to a learner's diagnosed knowledge gaps; continuously adapt personalization to the student's evolving needs across an interactive tutoring session",
  "goal_origin": "mixed:given-up-front-plus-emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Question generation and feedback in later turns must be calibrated against knowledge gaps diagnosed earlier in the same session, so a stale or wrong diagnosis of the learner cascades into miscalibrated later questions.",
  "n_goals": null,
  "tracking_demand": "The agent must track the learner's diagnosed knowledge gaps and profile, the history of prior interactions, and adapt difficulty calibration accordingly across a multi-turn tutoring dialogue.",
  "scoring": "LLM-judge-rubric",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "DeepTutor improves personalized metrics by 10.8% on average and strengthens general agentic reasoning across five backbone models by 29.4%; no explicit human-tutor baseline given.",
  "availability": "https://github.com/HKUDS/DeepTutor",
  "goal_span": "we introduce TutorBench, an interactive benchmark incorporating customized learner profiles grounded in university-level curricula across five domains. We further propose an LLM-based first-person interactive evaluation protocol that conducts assessments via a profile-driven student simulator.",
  "horizon_span": "an LLM-based first-person interactive evaluation protocol that conducts assessments via a profile-driven student simulator",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287916043",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291471936",
  "title": "CoCoBench: A Cooperative Coordination Benchmark for Embodied Multi-Agent Task Planning",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-08-28",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "CoCoBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "allocate tasks correctly among cooperating embodied agents; respect sequential-ordering constraints between agents' actions; respect mutual-exclusion constraints over shared objects/spaces; execute correct handoff coordination between agents; satisfy conjunctive multi-object-destination goals within executable household tasks",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The four coordination constructs (allocation, ordering, mutual exclusion, handoff) directly encode inter-agent dependencies -- e.g. one agent must open a drawer before another closes it only after all required objects are inside -- so violating one construct (duplicated work, ordering violation, resource contention, desynchronized handoff) causes coordination failure even if overall task success looks unaffected.",
  "n_goals": "897 oracle-validated instances across four coordination constructs",
  "tracking_demand": "The multi-agent system must track task allocation assignments, ordering/precedence constraints, mutual-exclusion locks on shared objects, and handoff status between agents, calibrated against a task-specific step budget H derived from oracle trajectories.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": "~18.4 mean executed action steps (best model, successful episodes); task-specific step budget H calibrated from oracle-validated trajectories",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per episode",
  "horizon_stated": "yes",
  "headline_result": "Coordination ability is highly construct-specific: strong overall performance does not imply balanced competence across coordination types (evaluated across 11 leading MLLMs); no single headline accuracy figure given in the abstract.",
  "availability": null,
  "goal_span": "CoCoBench contains 897 oracle-validated instances organized around four recurring coordination constructs: task allocation, sequential ordering, mutual exclusion, and handoff coordination. In addition to task success rate, CoCoBench provides construct-level scores that measure whether agents coordinate effectively.",
  "horizon_span": "the step budget is H, and episode succeeds when the final state satisfies every goal predicate ... GPT-5.6-sol achieves ... 18.4 mean steps per episode on successful attempts",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291471936",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287435959",
  "title": "DeCoNav: Dialog enhanced Long-Horizon Collaborative Vision-Language Navigation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 5,
  "publication_date": "2026-04-14",
  "months_since_pub": 5,
  "citations_per_month": 1.0,
  "artifact_name": "DeCoNavBench",
  "artifact_kind": "environment/simulator",
  "domain": "embodied-household",
  "goal_types": "achieve relay-style handoffs between two robots collaborating on a shared long-horizon navigation task; reach rendezvous points to exchange information/objects between robots; dynamically reassign and replan subgoals in response to new cross-agent evidence, uncertainty, or conflicts",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The two robots must operate on a synchronized shared world timeline, so handoff/rendezvous subgoals for one robot are contingent on the state and position of the other robot, and new evidence triggers replanning of the shared plan.",
  "n_goals": "1,213 tasks across 176 HM3D scenes",
  "tracking_demand": "Each robot must track its own navigation progress plus the other robot's evidence/uncertainty/conflicts communicated via event-triggered dialogue, in order to reassign subgoals and replan under synchronized execution.",
  "scoring": "other:both-success-rate-joint-binary - scored by 'both-success rate' (BSR), a joint criterion requiring both robots to succeed; abstract gives no indication of subgoal-level partial credit beyond this joint pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "DeCoNav improves the both-success rate (BSR) by 69.2% over the compared coordination approach; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "When informative events such as new evidence, uncertainty, or conflicts arise, dialogue is triggered to dynamically reassign subgoals and replan under synchronized execution. Implemented in DeCoNavBench with 1,213 tasks across 176 HM3D scenes",
  "horizon_span": "Long-horizon collaborative vision-language navigation (VLN) is critical for multi-robot systems to accomplish complex tasks beyond the capability of a single agent.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287435959",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290830120",
  "title": "Long-Horizon Embodied Decision-Making via Multimodal Memory Compression",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": true,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-08-02",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "DunphyBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "navigate through multiple embodied housing environments to gather evidence; integrate multimodal, multi-source input into coherent knowledge under partial observation; make a final housing decision aligned with multi-dimensional, partly implicit human preferences",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Evidence gathered while navigating one housing candidate must be integrated with evidence gathered from other candidates under partial observations in order to reach a single coherent, preference-aligned final decision.",
  "n_goals": null,
  "tracking_demand": "Agent must accumulate multimodal evidence across multiple candidate housing environments under partial observation and integrate it into a coherent decision aligned with multi-dimensional, partly implicit human preferences.",
  "scoring": "other:accuracy-gap-vs-human-performance",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "MeMento, a preference-conditioned memory compressor, improves VLM-driven agent accuracy by 7.18% while reducing memory usage by 85.38% versus the strongest baseline; a substantial but unquantified gap to human performance remains.",
  "availability": null,
  "goal_span": "we propose DunphyBench, a new benchmark for evaluating agents on long-horizon human-centered embodied decision-making, where the agent must navigate through multiple embodied housing environments and make decisions that align with multi-dimensional human preferences.",
  "horizon_span": "This shift requires agents to accumulate evidence over long horizons, interpret implicit user preferences, and compare multiple candidates under partial observations.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290830120",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291334607",
  "title": "Resilience Matters for Embodied Agents System: New Metrics, Systematic Evaluation, and Optimization",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-08-24",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "EAS resilience evaluation framework",
  "artifact_kind": "scenario-suite",
  "domain": "embodied-household",
  "goal_types": "complete a household task despite perturbations/unexpected disruptions during execution; recover, stabilize, and gracefully extend behavior after a perturbation rather than simply succeeding or failing outright; maintain low recovery cost and stability across iterative system updates",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "Recovery from an earlier perturbation affects the stability and extensibility of later behavior within the same episode; the framework measures recovery-cost differences between otherwise-successful episodes, showing that how a disruption was handled earlier changes later task-trajectory quality.",
  "n_goals": "3 resilience dimensions (Rebound, Stability, Graceful Extensibility) assessed across 400 household tasks with 10 EAS",
  "tracking_demand": "The evaluation layer must track the full execution trajectory of each household task under perturbation, not just final success/failure, to compute process-level resilience metrics like recovery cost and stability.",
  "scoring": "other:process-level-resilience-metrics. The framework explicitly moves beyond outcome-centric success/safety scores to trajectory-level metrics (Rebound, Stability, Graceful Extensibility); a continuous multi-dimensional process score rather than a binary or milestone rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Across 400 household tasks with 10 EAS, we reveal the process-level distinction hidden by outcome metrics, including recovery cost differences among successful episodes ($\\Delta C_{rec}=25.2$), increased instability and task-family degradation. Metrics-guided optimizations reduce recovery cost and increase stability, graceful extensibility completion, showing the diagnostic effect of resilience evaluation.",
  "horizon_span": "Across 400 household tasks with 10 EAS, we reveal the process-level distinction hidden by outcome metrics, including recovery cost differences among successful episodes ($\\Delta C_{rec}=25.2$)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291334607",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288257470",
  "title": "Ego2World: Compiling Egocentric Cooking Videos into Executable Worlds for Belief-State Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-05-13",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "Ego2World",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "plan under partial observation using a belief graph built from local observations; remember objects and track state changes in a hidden symbolic world graph; recover and replan when actions fail without observing the true world state; complete cooking-task objectives via correct sequences of graph-governed state transitions",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The agent's belief graph approximates a hidden true world graph; belief updates from earlier observations/actions constrain what the agent believes is possible later, and mismatches (e.g., missed state changes) cause replanning failures downstream, since actions fail when preconditions in the true (hidden) graph are not met.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain and update its own partial belief graph of the world (objects, state changes) using only local observations and execution feedback, separate from the simulator's hidden ground-truth world graph, and replan without directly observing true state.",
  "scoring": "other:physical-state-success-vs-action-overlap. The paper shows action-overlap scores overestimate physical-state success, and that persistent belief memory improves task completion while reducing repeated visual exploration -- distinct metrics (action overlap vs. true physical-state success) are compared rather than a single score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "During evaluation, the simulator maintains the hidden world graph, while the agent plans over its own partial belief graph using only local observations and execution feedback. This separation forces agents to update memory and replan without observing the true world state.",
  "horizon_span": "Embodied agents in household environments must plan under partial observation: they need to remember objects, track state changes, and recover when actions fail.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288257470",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290135330",
  "title": "ABot-AgentOS: A General Robotic Agent OS with Lifelong Multi-modal Memory",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-07-11",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "EmbodiedWorldBench (ABot-AgentOS)",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "navigate indoor/outdoor/hybrid scenes to reach targets; search for and identify specified objects; conduct NPC dialogue as part of a task; respond appropriately to dynamic events introduced mid-task; achieve trace-grounded, verifiable task completion across difficulty levels",
  "goal_origin": "mixed:given-up-front-tasks-with-environment-emitted-dynamic-events",
  "decomposition": "hierarchical",
  "interdependence": "Navigation, object search, and dialogue sub-tasks build on a persistent multi-modal memory graph, so later stages depend on facts (spatial, temporal, dialogue) accumulated in earlier stages, with edge-cloud collaboration and multi-stage verification enforcing checkpoints between stages.",
  "n_goals": "over 200 tasks across 16 scenes and four difficulty levels",
  "tracking_demand": "Agent must maintain a persistent, source-grounded multi-modal graph memory across dialogue, visual observations, spatial context, temporal relations, and task traces, continually updated and consulted for later decisions.",
  "scoring": "other:not-stated for EmbodiedWorldBench itself (task success/goal completion reported qualitatively as improved over baseline); on separate memory benchmarks (LoCoMo, OpenEQA, Mem-Gallery, NExT-QA) numeric accuracy scores are given (e.g. 87.5 on LoCoMo), which do reflect graded, checkpoint-style memory credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "ABot-AgentOS Static: 87.5 on LoCoMo, 59.9 on OpenEQA EM-EQA, 88.6 on Mem-Gallery, 76.5 Acc@All on NExT-QA; self-evolution further improves LoCoMo to 88.7, OpenEQA to 60.4, Mem-Gallery to 89.0 (no human-expert baseline given).",
  "availability": null,
  "goal_span": "we introduce EmbodiedWorldBench, an executable benchmark with 16 indoor, outdoor, and hybrid scenes, four difficulty levels, and over 200 tasks involving navigation, object search, NPC dialogue, dynamic events, and trace-grounded scoring",
  "horizon_span": "an executable benchmark with 16 indoor, outdoor, and hybrid scenes, four difficulty levels, and over 200 tasks",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290135330",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "288656195",
  "title": "TaskGround: Structured Executable Task Inference for Full-Scene Household Reasoning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-05-18",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "FullHome (via the TaskGround Ground-Infer-Execute framework)",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "identify task-relevant entities within a complete, cluttered household scene; recover intended task conditions implied by a situated (underspecified) household request; resolve ordering constraints among sub-actions from surrounding scene context; produce a grounded, skill-level action sequence that correctly executes the inferred task structure",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Correct execution depends on first correctly grounding the complete scene into a compact task-relevant slice and then correctly inferring executable task structure (entities, conditions, ordering) from that slice, so an error in grounding or inference propagates into an incorrect skill-level action sequence.",
  "n_goals": "400 household tasks (goal-oriented and process-constrained) across diverse home-scale environments",
  "tracking_demand": "The agent must track which entities, conditions, and ordering constraints it has inferred as task-relevant from a complete household scene, and maintain this executable task structure while compiling and executing a grounded skill-level action sequence.",
  "scoring": "other:task-success-rate-across-proprietary-and-open-weight-models",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "TaskGround improves task success rates by large margins across both proprietary and open-weight models on FullHome, making Qwen3.5-9B competitive with GPT-5 under direct complete-scene prompting while reducing input-token cost by up to 18x; no explicit human baseline is given (comparison is across models/methods).",
  "availability": null,
  "goal_span": "we introduce FullHome, a human-validated evaluation suite of 400 household tasks spanning diverse home-scale environments and both goal-oriented and process-constrained requirements. On FullHome, TaskGround improves task success rates by large margins across both proprietary and open-weight models.",
  "horizon_span": "an agent must infer executable task structure before producing a grounded skill-level action sequence",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288656195",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290648050",
  "title": "HumanCLAW: Can Vision-Language Models Act Through a Body?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 2,
  "publication_date": "2026-07-29",
  "months_since_pub": 2,
  "citations_per_month": 1.0,
  "artifact_name": "HumanCLAW-Bench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "find a specified target within an indoor scene; navigate to the target while avoiding obstacles/maintaining balance; interact with the target once reached; maintain embodied self-awareness of the body's own state (position, goal-reached status, collisions) throughout",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Interacting with the target is only meaningful once the agent has correctly navigated to it, which in turn depends on having correctly found/localized the target, so the three sub-goals are ordered; the paper's central finding is that models 'lose track of their own body,' i.e. failing to track body-state undermines all three sequential sub-goals.",
  "n_goals": "1,218 long-horizon, egocentric find-navigate-interact episodes across 41 indoor scenes",
  "tracking_demand": "The VLM must track where its body is, whether it has reached the goal, and whether it has hit an obstacle -- embodied self-awareness -- across each find-navigate-interact episode, since the paper finds this tracking, not target recognition, is the main bottleneck.",
  "scoring": "binary-final-success. The best model reaches only a 16.8% success rate; the abstract does not describe subgoal-checkpoint partial credit distinct from this overall per-episode success rate, though the three-phase (find/navigate/interact) structure could support such credit.",
  "horizon_value": "reported average of 78.5 steps per episode for baseline conditions",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per episode",
  "horizon_stated": "yes",
  "headline_result": "The best of nine evaluated state-of-the-art VLMs reaches only a 16.8% success rate on HumanCLAW-Bench; no human/expert baseline given in the abstract.",
  "availability": null,
  "goal_span": "we build HumanCLAW-Bench: 1,218 long-horizon, egocentric find-navigate-interact episodes across 41 indoor scenes. We test nine state-of-the-art VLMs and find that none solves the benchmark; the best model reaches only a 16.8% success rate. Recognizing the target is not the bottleneck. What current VLMs lack is embodied self-awareness: they lose track of their own body, failing to tell where it is, whether it has reached the goal, or whether it has hit an obstacle.",
  "horizon_span": "average steps in tables (e.g., 78.5 steps for baseline conditions)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290648050",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290331610",
  "title": "IMBench: A Benchmark for Intuitive Robotic Manipulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 0,
  "publication_date": "2026-07-17",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "IMBench",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "infer task-relevant physical structure via physical reasoning before acting; generate a feasible action sequence satisfying explicit task constraints (contact-rich manipulation, tool use, multi-stage dependencies); complete each of 35 manipulation tasks across scalable, diverse scenarios",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks have multi-stage dependencies (e.g., using a tool correctly depends on first inferring its physical affordances), so later manipulation stages are constrained by both physical structure inferred earlier and task-specific ordering requirements.",
  "n_goals": "35 tasks; 14K filtered trajectories",
  "tracking_demand": "Agent must track inferred physical structure (contacts, affordances), the current stage of a multi-stage manipulation plan, and constraint satisfaction across execution.",
  "scoring": "other:integrated-perception-reasoning-execution-score - evaluates intuitive manipulation as an integrated capability spanning perception, physical reasoning, action generation, and iterative execution; abstract frames results as a capability gap (VLMs reason but can't execute; VLA models struggle with constraints) rather than describing an explicit subgoal-checkpoint partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Our tasks require models to infer task-relevant physical structure and generate feasible action sequences under explicit constraints, including contact-rich manipulation, tool use, and multi-stage dependencies.",
  "horizon_span": "We introduce a benchmark of 35 tasks, 14K filtered trajectories, and scalable tools for generating diverse scenarios.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290331610",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284860982",
  "title": "Explore with Long-term Memory: A Benchmark and Multimodal LLM-based Reinforcement Learning Framework for Embodied Exploration",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 10,
  "publication_date": "2026-01-11",
  "months_since_pub": 8,
  "citations_per_month": 1.25,
  "artifact_name": "LMEE-Bench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "perform multi-goal navigation across an embodied environment; answer memory-based questions using episodic memory accumulated during exploration; unify exploratory cognition with decision-making to support lifelong learning; proactively query memory and select exploration frontiers next",
  "goal_origin": "mixed:navigation-goals-given-up-front-exploration-strategy-self-generated",
  "decomposition": "sequential-chain",
  "interdependence": "Memory-based question answering depends on episodic memory accumulated during earlier exploration and navigation, so correct multi-goal navigation and correct recall/QA both depend on how well the agent explored and stored experience in previous steps of the same episode.",
  "n_goals": null,
  "tracking_demand": "The agent must track its accumulated episodic memory, current exploration frontier, and multiple concurrent navigation goals across a long-horizon embodied exploration episode.",
  "scoring": "other:joint-navigation-and-memory-QA-evaluation. The benchmark incorporates multi-goal navigation and memory-based question answering together to evaluate both the process (exploration) and outcome (task completion) of embodied exploration, rather than a single final binary score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://wangsen99.github.io/papers/lmee/",
  "goal_span": "We further construct a corresponding dataset and benchmark, LMEE-Bench, incorporating multi-goal navigation and memory-based question answering to comprehensively evaluate both the process and outcome of embodied exploration.",
  "horizon_span": "An ideal embodied agent should possess lifelong learning capabilities to handle long-horizon and complex tasks, enabling continuous operation in general environments.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284860982",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288254096",
  "title": "When Robots Do the Chores: A Benchmark and Agent for Long-Horizon Household Task Execution",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 2,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "LongAct",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "understand a free-form (non-templated) household instruction and decompose it into the right sequence of sub-tasks; manage dependencies among sub-tasks (ordering, shared resources) via a DAG-based plan; maintain persistent memory (spatial and episodic) across a long-horizon household task; adapt the plan reflectively as execution proceeds",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "LongAct isolates high-level cognitive capabilities such as instruction understanding, dependency management, memory maintenance, and adaptive planning: HoloMind's DAG-based hierarchical planner and spatial/episodic memory explicitly track how earlier sub-task outcomes and world state constrain feasible later sub-tasks in the same free-form household task.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain persistent spatial and episodic memory of what it has already done and observed in the household, use this to keep dependency-aware track of remaining sub-tasks, and adaptively replan as new information arises over the long-horizon task.",
  "scoring": "other:goal-completion-and-full-task-success-rate. Even top models achieve only 59% goal completion and 16% full-task success, i.e. the paper explicitly reports partial (per-goal) completion separately from strict full-task success, a form of subgoal-level credit distinct from binary pass/fail.",
  "horizon_value": "human executors typically require 500+ steps; VLM-based agents often exceed 2,000 steps to complete the same long-horizon household task",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/episode",
  "horizon_stated": "yes",
  "headline_result": "Even the top evaluated models (GPT-5, Qwen3-VL-based HoloMind) achieve only 59% goal completion and 16% full-task success on LongAct; no human/expert baseline given in the abstract.",
  "availability": null,
  "goal_span": "We introduce LongAct, a benchmark designed to evaluate planning-level autonomy in long-horizon household tasks specified through free-form instructions. By abstracting away embodiment-specific low-level control, LongAct isolates high-level cognitive capabilities such as instruction understanding, dependency management, memory maintenance, and adaptive planning.",
  "horizon_span": "While human executors typically require more than 500 steps to complete a task, VLM-based agents often exceed 2,000 steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288254096",
  "provenance": "asta-find,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289691087",
  "title": "MECoBench: A Systematic Study of Multimodal Agent Collaboration in Embodied Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 2,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "MECoBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "complete embodied tasks under two cooperation structures and three collaboration modes; coordinate communication among multiple multimodal agents in a visually grounded environment; remain robust to noisy priors and exploration conditions via collaboration",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Collaborative gains depend on balancing communication benefits against coordination complexity, and the best collaboration mode depends jointly on team size and model capability, so goal achievement across agents is mutually dependent rather than independent.",
  "n_goals": null,
  "tracking_demand": "Agents must track and communicate task-relevant state among team members across two cooperation structures and three collaboration modes to complete embodied tasks under noisy priors and exploration conditions.",
  "scoring": "other:qualitative-cross-condition-findings",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/q-i-n-g/MECoBench",
  "goal_span": "we introduce MECoBench, a multimodal embodied cooperation benchmark with an evaluation platform spanning diverse real-world tasks, two cooperation structures, and three collaboration modes... (i) Collaboration generally improves embodied task completion, but its benefits depend on balancing collaborative gains against coordination complexity.",
  "horizon_span": "an evaluation platform spanning diverse real-world tasks, two cooperation structures, and three collaboration modes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289691087",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289690923",
  "title": "MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 0,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "MultiUAV-Plat Benchmark",
  "artifact_kind": "environment/simulator",
  "domain": "robotics-sim",
  "goal_types": "assign UAVs to targets correctly under partial observability; perform area search coverage objectives; perform area assignment and patrol objectives across multiple UAVs; satisfy each of the mission's validation checks (mean 6.26/task) via correct multi-vehicle coordination",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Multiple UAVs share airspace and mission resources under partial observability and role-based information access, so target-assignment and area-coverage decisions for one UAV constrain what remains available for the others within the same mission session.",
  "n_goals": "75 mission sessions, 1,500 natural-language tasks, 9,396 validation checks (median 4, mean 6.26 checks/task)",
  "tracking_demand": "The framework (Agent4Drone) must track memory, observation, task understanding, planning, execution, and verification state across each mission, checking role-based information access and validation logic per task within a session.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Agent4Drone achieves 57.9% task pass rate vs. ReAct baseline's 30.6%; reduces total failed task rate from 32.4% to 12.9%; no human baseline given.",
  "availability": null,
  "goal_span": "The MultiUAV-Plat Benchmark contains 75 mission sessions, 1500 natural-language tasks, and 9396 validation checks across target assignment, area search, and area assignment and patrol scenarios. ... Agent4Drone achieves a 57.9% task pass rate, a 74.6% average task check pass rate, and a 72.0% global check pass rate",
  "horizon_span": "check counts per task (median of 4, mean of 6.26)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289690923",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289668977",
  "title": "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-06-26",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "PARTNR-Dialog",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "complete a shared household task cooperatively with a partner agent; communicate with the partner (e.g., 'talk when necessary') to align on task objectives; align actions with the partner's behavior and the environment state to avoid inefficient or conflicting actions; apply learned high-level behavioral laws (e.g., 'wait for partner') during planning",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Success requires the agent's actions to remain aligned with both its partner's behavior and evolving task/environment state; misaligned communication or action timing (e.g., not waiting for a partner) causes inefficient cooperation, so earlier communicative/behavioral choices constrain which later joint actions succeed.",
  "n_goals": null,
  "tracking_demand": "The agent must track its partner's current actions/state, the shared household task's progress, and which high-level behavioral laws (e.g., 'talk when necessary,' 'wait for partner') apply at each point in the cooperative plan.",
  "scoring": "other:cooperative-task-success-rate-with-efficiency. The paper reports task success-rate improvements (4.5% on PARTNR-Dialog, 6.8% on TDW-MAT) alongside cooperative-efficiency gains, rather than a single binary score reported without efficiency context.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "LLawCo achieves average success-rate improvements of 4.5% on PARTNR-Dialog and 6.8% on TDW-MAT over state-of-the-art open-source communicative agent frameworks, across four backbone LLMs; no human/expert baseline given.",
  "availability": "https://www.merl.com/research/highlights/LLawCo",
  "goal_span": "we introduce PARTNR-Dialog, a large-scale multi-agent communicative and cooperative planning benchmark built on the PARTNR environment. Experiments on existing tasks and our new benchmark demonstrate significant improvements in cooperative efficiency and task success rates.",
  "horizon_span": "Embodied agents operating in decentralized and partially observable environments have attracted growing attention in recent years.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289668977",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285270000",
  "title": "PLanAR: Planning-Language-Grounded Agentic Reasoning for Robot Manipulation",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 3,
  "publication_date": "2026-02-02",
  "months_since_pub": 7,
  "citations_per_month": 0.43,
  "artifact_name": "PLanAR",
  "artifact_kind": "scenario-suite",
  "domain": "robotics-sim",
  "goal_types": "represent and update object-predicate scene states as manipulation proceeds; select and sequence action schemas (with preconditions/effects) to satisfy an open-vocabulary manipulation goal; detect execution failures via stepwise symbolic-effect verification and replan; complete long-horizon kitchen workflows composed of many chained manipulation sub-goals",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Action schemas have explicit preconditions and effects, so later actions in a plan are only valid once earlier actions' expected symbolic effects have been verified via onboard observation; a failed precondition triggers replanning, directly coupling sub-goal order.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain a symbolic scene-state representation (object predicates), verify after each action whether expected effects were achieved, and update/replan when execution deviates from the expected symbolic plan across a long-horizon kitchen workflow.",
  "scoring": "other:qualitative-real-world-task-success-across-embodiments",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "PLanAR uses a planning-language interface to define the VLM reasoning space: object predicates represent scene states, action schemas specify robot skills with preconditions and effects, and symbolic plans provide executable intermediate representations. This interface enables stepwise verification: after each action, PLanAR uses onboard observations to check whether the expected symbolic effects have been achieved, allowing the VLM-based agent to update task states, detect failures, and replan when execution deviates from expectation.",
  "horizon_span": "Across robot embodiments, VLM backends, and tasks including stacking, crossword solving, and long-horizon kitchen workflows, PLanAR demonstrates strong real-world capability",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285270000",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288861916",
  "title": "RescueBench: Can Embodied Agents Save Lives in the Wild ?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-06-01",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "RescueBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "explore an unfamiliar environment under multimodal uncertainty to locate a target (multimodal exploration stage); physically rescue/reach the identified target (target rescue stage); navigate back using retained spatial memory (memory-guided return stage); complete a final handoff of the rescued target/information (final handoff stage)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each stage depends on successful completion and correct information retention from the prior stage -- e.g. memory-guided return requires spatial memory acquired during earlier exploration -- so failures at one stage propagate and prevent later stages from succeeding.",
  "n_goals": "four sequential stages per episode, evaluated across five progressive difficulty levels",
  "tracking_demand": "The agent must retain spatial memory of the environment (for the return stage), track clue ambiguity/target identification state, and manage per-level time budgets (180-300 seconds) across the four-stage pipeline.",
  "scoring": "subgoal-checkpoint-partial-credit -- stage-level evaluation explicitly scores each of the four stages (exploration, rescue, memory-guided return, handoff) rather than only final task success, enabling diagnosis of how failures propagate across stages.",
  "horizon_value": "episodes are governed by level-dependent time limits: L1-L2: 180 seconds, L3: 240 seconds, L4-L5: 300 seconds, with no step cap",
  "horizon_unit": "other:wall-clock-seconds",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per difficulty-level time budget)",
  "horizon_stated": "yes",
  "headline_result": "No baseline completes the full task at the greatest (five-level) difficulty, versus human players and an oracle reference; a specific numeric agent-vs-human gap is not given in the abstract.",
  "availability": "https://github.com/wukui-muc/RescueBench",
  "goal_span": "We introduce RescueBench, a photo-realistic diagnostic benchmark that instantiates SAR as a four-stage pipeline: multimodal exploration, target rescue, memory-guided return, and final handoff. By combining sequential task composition with stage-level evaluation, RescueBench enables analysis of how exploration and memory failures propagate through embodied rescue workflows.",
  "horizon_span": "Resource budgets: episodes are governed by level-dependent time limits (L1-L2: 180 s, L3: 240 s, L4-L5: 300 s) with no step cap.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288861916",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290984464",
  "title": "Compiling and Benchmarking Task-State Horizons for Embodied Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 0,
  "publication_date": "2026-08-08",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "RoboGraph",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "track the evolving span of task-relevant state transitions (task-state horizon, TSH) induced by both exploration and environmental dynamics; correctly execute high-level plans as task-relevant world state changes, including due to unexpected failures/interventions; complete each of 588 episodes across 84 scenes with varying task-state horizons",
  "goal_origin": "mixed:task-given-up-front-state-transitions-emitted-by-environment",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Task-state horizon (TSH) is defined as the span of task-relevant state transitions an agent must track, and RoboGraph compiles these from spatial and temporal causal dependencies including unexpected failures and interventions, so later plan decisions depend on correctly tracking state transitions caused earlier, whether by the agent's own exploration or external dynamics.",
  "n_goals": "588 episodes across 84 scenes with varying task-state horizons",
  "tracking_demand": "The agent must maintain, explore, and update task-relevant world state over the course of a long-horizon rollout, since performance is explicitly measured as a function of the task-state horizon (TSH) -- the span of state transitions it must track.",
  "scoring": "other:performance-as-a-function-of-task-state-horizon(TSH). The benchmark's central contribution is measuring how agent performance varies with a graded difficulty axis (TSH) rather than a single undifferentiated success rate, an explicit graded/continuous difficulty-conditioned scoring design.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We define the span of task-relevant state transitions that an agent must track as task-state horizon (TSH)... Building on RoboGraph, we release a benchmark comprising 588 episodes across 84 scenes with varying TSHs.",
  "horizon_span": "We define the span of task-relevant state transitions that an agent must track as task-state horizon (TSH).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290984464",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288861040",
  "title": "SMH-Bench: Benchmarking LLM Agents for Environment-Grounded Reasoning and Action in Smart Homes",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 2,
  "publication_date": "2026-06-01",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "SMH-Bench (built on HomeEnv)",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "execute explicit device control and query commands across a smart home with up to 135 devices; schedule automation tasks that must fire correctly over time; correctly handle ambiguous user instructions; personalize reasoning to user intent/preferences as home complexity increases",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Within a single complex-home task, multiple devices (up to 135) share physical/state relationships, so automation scheduling or ambiguity resolution for one device's state can depend on the current state of other devices in the same multi-room home.",
  "n_goals": "1,100 tasks spanning 7 categories and 22 fine-grained subcategories, stratified across simple/medium/complex homes (up to 135 devices)",
  "tracking_demand": "Agent must track the state of many concurrent devices across a multi-room home, especially for automation-scheduling tasks that must persist and correctly re-trigger over time, and must resolve ambiguous instructions against current home/device state.",
  "scoring": "binary-final-success - built upon an executable and verifiable simulator (HomeEnv); abstract reports capability weaknesses (automation scheduling, ambiguity handling, personalized reasoning) as home complexity increases, but does not describe an explicit subgoal-checkpoint partial-credit scheme beyond per-task pass/fail implied by 'verifiable' tasks.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Frontier LLMs achieve strong performance on explicit control and query tasks but exhibit significant weaknesses in automation task scheduling, ambiguity handling, and personalized reasoning, especially as home complexity increases; no specific numeric top score or human baseline is given.",
  "availability": null,
  "goal_span": "Built upon HomeEnv, an executable and verifiable smart-home simulator, SMH-Bench contains 1,100 high-quality tasks spanning 7 categories and 22 fine-grained subcategories. It further stratifies tasks across simple, medium and complex homes, ranging from small apartments to dense multi-room environments with 135 devices.",
  "horizon_span": "It further stratifies tasks across simple, medium and complex homes, ranging from small apartments to dense multi-room environments with 135 devices.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288861040",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289098608",
  "title": "SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 2,
  "publication_date": "2026-06-08",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "SpatialWorld",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "actively gather egocentric visual evidence under partial observability to resolve a task; complete household, travel, or social-collaboration tasks requiring interactive spatial reasoning; express decisions via a unified text-based action interface across eight heterogeneous simulator backends",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Because agents act under 'vision-only partial observability', later action choices depend on visual evidence actively gathered in earlier steps, and a human-validated reference trajectory plus terminal-state verifier ties the whole task to a single connected sequence of exploration-then-decision steps.",
  "n_goals": "760 human-annotated tasks",
  "tracking_demand": "Agent must track what it has and has not yet observed (partial observability), accumulate egocentric visual evidence over the course of the task, and reconcile this with a reference trajectory/terminal-state verifier.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports a continuous 'task success rate (TSR)' metric (e.g. 17.4% for GPT-5) alongside a separate execution-efficiency measure, but does not describe explicit subgoal-checkpoint partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Strongest model GPT-5 achieves an average task success rate (TSR) of only 17.4%; leading open-source model Qwen-3.5 reaches 14.1% (no human baseline given).",
  "availability": null,
  "goal_span": "Agents must solve tasks under vision-only partial observability, actively gathering egocentric visual evidence and expressing decisions via a unified, text-based action interface native to MLLMs.",
  "horizon_span": "SpatialWorld features 760 human-annotated tasks across diverse domains (e.g., household routines, travel, social collaboration)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289098608",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "289941520",
  "title": "TypeGo: An OS Runtime for Embodied Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:physical-embodied-robot-runtime",
  "citation_count": 0,
  "publication_date": "2026-07-06",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "TypeGo (Kalos prototype)",
  "artifact_kind": "environment/simulator",
  "domain": "other:physical-embodied-robot-runtime",
  "goal_types": "execute multiple concurrent per-task processes/goals on a shared physical robot body without conflicting over physical subsystems; preempt, resume, or replace a lower-priority task/goal when a new goal arrives; maintain fast first-action responsiveness while longer-horizon planning continues asynchronously in the background",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "Concurrent per-task processes/goals all compete for the same shared physical subsystems (the robot's body), so the Skill Kernel and scheduler must arbitrate, preempt, and resume/replace tasks to avoid physically conflicting actions, meaning one goal's execution can block or override another's until resolved.",
  "n_goals": null,
  "tracking_demand": "The runtime must track which per-task processes currently hold which physical subsystems, their priority/preemption state, and pending speculative skill-streaming actions, to arbitrate concurrent goals in real time.",
  "scoring": "other:latency-and-concurrency-efficiency. The abstract reports it 'cuts per-step delay by 50%... and time-to-first-action by 73%... while admitting concurrent tasks at low scheduling overhead' -- system-level efficiency/responsiveness metrics rather than a task-success or milestone rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "the Skill Kernel arbitrates typed physical subsystems among concurrent per-task processes, a scheduler preempts them and resumes or replaces each by source, and speculative skill streaming hides LLM latency behind ongoing motion... it cuts per-step delay by 50% over step-by-step planning and time-to-first-action by 73% over monolithic planning, while admitting concurrent tasks at low scheduling overhead.",
  "horizon_span": "our prototype of Kalos, a Unitree Go2 quadruped, provides preliminary evidence for the design: in our current task suite, it cuts per-step delay by 50% over step-by-step planning and time-to-first-action by 73% over monolithic planning",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289941520",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290377398",
  "title": "UniETP: Unifying Environments for Generalizable Embodied Task Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-07-20",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "UniETP",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "execute a sequence of atomic actions within an interactive environment to complete a user-specified task; handle varying task-logic complexity; handle varying instance-grounding complexity; handle varying instruction-understanding complexity, across four unified simulators (AI2-THOR, VirtualHome, Habitat, BEHAVIOR)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Task difficulty varies along three coupled dimensions (task logic, instance grounding, instruction understanding), so correctly executing later atomic actions in the sequence depends on correctly resolving earlier grounding and instruction-understanding decisions within the unified action space.",
  "n_goals": null,
  "tracking_demand": "The agent must track its progress executing a sequence of atomic actions toward a user-specified goal within a unified observation/action space, while contending with varying levels of task-logic, instance-grounding, and instruction-understanding difficulty across four different underlying simulators.",
  "scoring": "other:standardized-cross-simulator-evaluation-system. The paper builds an evaluation system supporting complicated task goals across four previously siloed simulators, enabling comprehensive model comparison, but the abstract does not describe an explicit subgoal-checkpoint partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/woyut/UniETP",
  "goal_span": "This paper focuses on the problem of Embodied Task Planning, where an agent is required to execute a sequence of atomic actions within an interactive environment to complete a user-specified task... it formalizes all the simulators into a consistent observation and action space, and builds an evaluation system to support complicated task goal.",
  "horizon_span": "an agent is required to execute a sequence of atomic actions within an interactive environment to complete a user-specified task",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290377398",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291544646",
  "title": "Towards Generalizable Visually Grounded Exploration of Household Devices",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": "2026-09-01",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "VGEBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "form a hypothesis about how to operate a novel household device from visual cues alone (no manual); test the hypothesis via physical interaction and interpret feedback; refine/correct the hypothesis and action based on observed feedback (Hypothesis-Interaction-Refinement loop) until the device is successfully operated",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Each refinement step depends on the outcome of the previous interaction attempt (the physical feedback observed), so the agent must revise its hypothesis and visual-affordance interpretation based on cumulative interaction history rather than treat each attempt independently.",
  "n_goals": "4,948 single-turn and 10,005 multi-turn instructions constructed for evaluation; no fixed number of refinement iterations per instance is stated",
  "tracking_demand": "The agent must track its current hypothesis about the device's operation, the history of physical feedback received from prior interaction attempts, and dynamically-calculated interaction budgets based on task complexity, maintaining long-horizon state across the exploration loop.",
  "scoring": "other:dynamic-interaction-budget-completion -- success is judged by whether the agent achieves the functional goal within a task-complexity-dependent interaction budget; the abstract does not describe explicit intermediate-step partial credit beyond overall task success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "This framework simulates multi-turn interaction loops, compelling agents to achieve goals by active visual perception and feedback-driven correction. Experimental results demonstrate that existing VLMs face significant challenges in translating semantic knowledge into physical execution and maintaining long-horizon state tracking.",
  "horizon_span": "In total, we constructed 4,948 single-turn and 10,005 multi-turn instructions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291544646",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289626156",
  "title": "WorldLines: Benchmarking and Modeling Long-Horizon Stateful Embodied Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 1,
  "publication_date": "2026-06-17",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "WorldLines",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "answer Memory QA questions correctly using long-term household interaction history; produce correct Embodied Task Plans grounded in remembered user routines and past world/device states; maintain visibility-aware (partially observable) memory across a temporally extended household trace",
  "goal_origin": "given-up-front",
  "decomposition": "other:evidence-linked-samples-drawn-from-a-shared-temporal-trace",
  "interdependence": "Memory QA and Embodied Task Planning samples are both grounded in the same temporally extended household trace, so correct answers/plans require recalling routines, prior dialogues, actions, and object/device state changes established earlier in that shared trace under partial observability.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain visibility-aware memory of user routines, world states, past actions/dialogues, and object/device state changes across long temporally extended household traces to answer Memory QA and produce embodied plans.",
  "scoring": "other:evidence-linked-sample-accuracy. No explicit subgoal-checkpoint partial-credit scheme is described; each converted Memory QA / Task Planning sample is evaluated against its own evidence-linked ground truth.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "It constructs temporally extended household traces with dialogues, actions, execution feedback, object and device state changes, and converts them into evidence-linked samples for Memory QA and Embodied Task Planning.",
  "horizon_span": "It constructs temporally extended household traces with dialogues, actions, execution feedback, object and device state changes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289626156",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278959847",
  "title": "3DLLM-Mem: Long-Term Spatial-Temporal Memory for Embodied 3D Large Language Model",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 29,
  "publication_date": "2025-05-28",
  "months_since_pub": 16,
  "citations_per_month": 1.81,
  "artifact_name": "3DMem-Bench (via 3DLLM-Mem)",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "correctly recall and apply past spatial-temporal observations (episodic memory) to complete embodied tasks in multi-room 3D environments; answer questions and produce captions grounded in accumulated long-term memory of the 3D scene; focus on task-relevant information while maintaining memory efficiency across complex, long-horizon environments",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Correctly completing a current task/question depends on correctly recalling and fusing spatial-temporal features from past observations stored in episodic memory, so the agent's current 'working memory' queries are directly coupled to what was retained from earlier parts of the same long trajectory.",
  "n_goals": "over 26,000 trajectories and 2,892 embodied tasks (plus question-answering and captioning instances), across multi-room 3D scenes with varying difficulty levels",
  "tracking_demand": "The agent must maintain and selectively query an episodic memory of past spatial and temporal observations across long-horizon, multi-room 3D trajectories, focusing on task-relevant information while remaining memory-efficient.",
  "scoring": "other:success-rate-comparison -- 3DLLM-Mem outperforms the strongest baselines by 16.5% in success rate on 3DMem-Bench's most challenging in-the-wild embodied tasks, a continuous success-rate comparison rather than a stated subgoal-checkpoint partial-credit rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "3DLLM-Mem outperforms the strongest baselines by 16.5% in success rate on 3DMem-Bench's most challenging in-the-wild embodied tasks; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "we first introduce 3DMem-Bench, a comprehensive benchmark comprising over 26,000 trajectories and 2,892 embodied tasks, question-answering and captioning, designed to evaluate an agent's ability to reason over long-term memory in 3D environments... Our approach allows the agent to focus on task-relevant information while maintaining memory efficiency in complex, long-horizon environments.",
  "horizon_span": "we first introduce 3DMem-Bench, a comprehensive benchmark comprising over 26,000 trajectories and 2,892 embodied tasks, question-answering and captioning, designed to evaluate an agent's ability to reason over long-term memory in 3D environments.",
  "extraction_confidence": 2,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278959847",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283250994",
  "title": "ArtiBench and ArtiBrain: Benchmarking Generalizable Vision-Language Articulated Object Manipulation",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 0,
  "publication_date": "2025-11-25",
  "months_since_pub": 10,
  "citations_per_month": 0.0,
  "artifact_name": "ArtiBench (with ArtiBrain)",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "manipulate articulated objects (kitchen, storage, office, tool appliances) correctly across parts/instances/categories; decompose and validate a sequence of sub-goals for a long-horizon, multi-object manipulation task; maintain physical consistency across the multi-step interaction",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "ArtiBrain's VLM-based Task Reasoner decomposes and validates subgoals hierarchically before a Hybrid Controller executes them, so downstream low-level actions depend on upstream subgoal validation, and successful manipulation of later parts/objects can depend on affordances learned from earlier ones via the Affordance Memory Bank.",
  "n_goals": null,
  "tracking_demand": "System must track subgoal validation state (via the Task Reasoner), accumulated part-level affordances in the Affordance Memory Bank, and physical consistency across a five-level benchmark spanning cross-part/instance/category variation to long-horizon multi-object tasks.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports 'ArtiBrain significantly outperforms state-of-the-art multimodal and diffusion-based methods in robustness and generalization' but does not describe a subgoal-checkpoint credit scheme explicitly (though the Task Reasoner does 'decompose and validate subgoals', implying some internal subgoal-level gating).",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "ArtiBrain 'significantly outperforms state-of-the-art multimodal and diffusion-based methods in robustness and generalization' (no specific numeric score or human baseline given in the abstract).",
  "availability": "Code and dataset will be released upon acceptance (no URL given yet).",
  "goal_span": "ArtiBench enables structured evaluation from cross-part and cross-instance variation to long-horizon multi-object tasks, revealing the core generalization challenges of articulated object manipulation",
  "horizon_span": "ArtiBench, a five-level benchmark covering kitchen, storage, office, and tool environments",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283250994",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "283466756",
  "title": "Benchmark for Planning and Control with Large Language Model Agents: Blocksworld with Model Context Protocol",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 2,
  "publication_date": "2025-12-03",
  "months_since_pub": 9,
  "citations_per_month": 0.22,
  "artifact_name": "Blocksworld-MCP benchmark",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "reach one target block-configuration goal state via a sequence of pick/put/stack/unstack actions; satisfy the goal under increasing plan-length complexity categories (step-2 through step-12); generalize plan validity/optimality across diverse agent architectures connected via a standardized MCP tool interface",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each block-move action changes the shared table/stack configuration, so later actions' validity depends on the exact configuration left by earlier actions in the same plan; achieving the target configuration requires the whole ordered action sequence to be jointly correct, not just each action in isolation.",
  "n_goals": "five complexity categories; scenarios span step-2 through step-12 optimal-plan-length categories (45 step-2, 84 step-4, 152 step-6, 151 step-8, 112 step-10, 46 step-12 scenarios), each involving up to five blocks",
  "tracking_demand": "The agent must track the current block/stack configuration as pick/put/stack/unstack actions are executed, correctly planning toward the target configuration across plans of up to 12 optimal steps, potentially under constrained block size or partial observability.",
  "scoring": "other:plan-validity-and-optimality -- core metrics include plan validity rate, plan optimality, grounding accuracy (for vision-language tasks), execution time, and resource use, multiple distinct metrics rather than a single binary success/failure.",
  "horizon_value": "scenarios span step-2 through step-12 optimal-plan-length categories (45 step-2, 84 step-4, 152 step-6, 151 step-8, 112 step-10, 46 step-12 scenarios), each involving up to five blocks",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task (per scenario's optimal plan length)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce a benchmark with an executable simulation environment representing the Blocksworld problem providing five complexity categories. By integrating the Model Context Protocol (MCP) as a standardized tool interface, diverse agent architectures can be connected to and evaluated against the benchmark without implementation-specific modifications. A single-agent implementation demonstrates the benchmark's applicability, establishing quantitative metrics for comparison of LLM-based planning and execution approaches.",
  "horizon_span": "the dataset includes 45 step-2, 84 step-4, 152 step-6, 151 step-8, 112 step-10, and 46 step-12 scenarios, each involving up to five blocks",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283466756",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "280526629",
  "title": "CookBench: A Long-Horizon Embodied Planning Benchmark for Complex Cooking Scenarios",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 13,
  "publication_date": "2025-08-05",
  "months_since_pub": 13,
  "citations_per_month": 1.0,
  "artifact_name": "CookBench",
  "artifact_kind": "environment/simulator",
  "domain": "embodied-household",
  "goal_types": "accurately parse a user's complex cooking intent (Intention Recognition); execute the identified cooking goal through a long-horizon, fine-grained sequence of physical actions (Embodied Interaction); correctly use both macro-level operations (placing orders, purchasing ingredients) and fine-grained embodied physical actions",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Embodied Interaction success is conditioned on first correctly completing Intention Recognition, since executing the wrong cooking goal makes the subsequent long sequence of physical actions incorrect regardless of execution quality; within Embodied Interaction, macro operations (e.g., purchasing ingredients) must precede the fine-grained physical actions that depend on those ingredients being available.",
  "n_goals": null,
  "tracking_demand": "The agent must track the parsed cooking intent, its progress through a long-horizon fine-grained sequence of physical actions, and the state of ingredients/tools obtained via macro-level operations, across a two-stage cooking task.",
  "scoring": "other:in-depth-qualitative-shortcomings-analysis-of-closed-source-LLM-VLM",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "The core task in CookBench is designed as a two-stage process. First, in Intention Recognition, an agent needs to accurately parse a user's complex intent. Second, in Embodied Interaction, the agent should execute the identified cooking goal through a long-horizon, fine-grained sequence of physical actions.",
  "horizon_span": "the agent should execute the identified cooking goal through a long-horizon, fine-grained sequence of physical actions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280526629",
  "provenance": "asta-find,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "278237692",
  "title": "DeCo: Task Decomposition and Skill Composition for Zero-Shot Generalization in Long-Horizon 3D Manipulation",
  "year": 2025,
  "venue": "IEEE Robotics and Automation Letters",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 16,
  "publication_date": "2025-05-01",
  "months_since_pub": 16,
  "citations_per_month": 1.0,
  "artifact_name": "DeCoBench",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "retrieve and chain reusable atomic manipulation skills to complete a compositional long-horizon 3D manipulation task; generalize zero-shot to novel task compositions not seen during training; execute smooth, collision-free transitions between chained skills",
  "goal_origin": "mixed:high-level-instruction-given-up-front-subgoal-decomposition-self-generated",
  "decomposition": "sequential-chain",
  "interdependence": "Skills are chained sequentially such that the end-state of one atomic skill becomes the start-state for the next, and a spatially-aware chaining module must ensure collision-free transitions, so a poor transition invalidates subsequent skill execution.",
  "n_goals": "trained on only 6 atomic tasks; evaluated on 12 novel simulated tasks and 9 novel real-world tasks",
  "tracking_demand": "The system must track the currently retrieved skill, the object/gripper state at each transition point, and which atomic subtask in the composed sequence is currently active to ensure valid chaining.",
  "scoring": "other:per-task-success-rate-with-skill-composition-analysis. The abstract reports success-rate improvements (66.67%, 21.53%, 57.92%, 53.33%) on composed tasks relative to base imitation-learning models rather than an explicit per-subgoal checkpoint scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "DeCo improves success rate of RVT-2, 3DDA, and ARP by 66.67%, 21.53%, and 57.92% on 12 novel simulated tasks; in real-world tests (trained on 6 atomic tasks) completes 9 novel tasks zero-shot with a 53.33% improvement over baseline; no human/expert baseline given.",
  "availability": null,
  "goal_span": "DeCo decomposes IL demonstrations into modular atomic tasks based on gripper-object interactions, creating a dataset that enables models to learn reusable skills. At inference, DeCo uses a vision-language model (VLM) to parse high-level instructions, retrieve relevant skills, and dynamically schedule their execution. A spatially-aware skill-chaining module ensures smooth, collision-free transitions between skills.",
  "horizon_span": "trained on only 6 atomic tasks, completes 9 novel tasks in zero-shot",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278237692",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284077881",
  "title": "DeliveryBench: Can Agents Earn Profit in Real World?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 5,
  "publication_date": "2025-12-22",
  "months_since_pub": 9,
  "citations_per_month": 0.56,
  "artifact_name": "DeliveryBench",
  "artifact_kind": "environment/simulator",
  "domain": "robotics-sim",
  "goal_types": "maximize net profit over the course of an operating shift by choosing which deliveries to accept/complete; meet each accepted delivery's deadline; manage limited resources (transportation expense, vehicle battery) across the shift; interact appropriately with other couriers and customers as needed",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "All accepted deliveries share the same limited transportation budget and vehicle battery, and deadlines create timing conflicts, so accepting/completing one delivery can consume resources or time needed for another, and irreversible choices (e.g. depleting battery far from a charging point) can foreclose future deliveries.",
  "n_goals": "long-horizon objectives instantiated per shift across nine procedurally generated cities; typically >100 action steps and several in-game hours per episode",
  "tracking_demand": "The agent must track remaining budget/vehicle battery, delivery deadlines for all currently accepted orders, and its evolving location within a procedurally generated 3D city, over an episode lasting several in-game hours and typically more than 100 action steps.",
  "scoring": "continuous-reward -- performance is measured via net profit accumulated over the shift, compared against human players who substantially outperform the agents; the abstract does not describe a subgoal-checkpoint partial-credit rubric beyond this continuous profit measure.",
  "horizon_value": "episodes support long-horizon tasks spanning several in-game hours and typically more than 100 action steps",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode",
  "horizon_stated": "yes",
  "headline_result": "Substantial performance gap to human players is found across nine cities; agents are short-sighted and frequently break basic commonsense constraints; distinct model 'personalities' are observed (e.g. adventurous GPT-5 vs. conservative Claude); no single numeric best-agent-vs-human gap is given in the abstract.",
  "availability": "https://deliverybench.github.io",
  "goal_span": "Food couriers naturally operate under long-horizon objectives (maximizing net profit over hours) while managing diverse constraints, e.g., delivery deadline, transportation expense, vehicle battery, and necessary interactions with other couriers and customers.",
  "horizon_span": "DeliveryBench supports long-horizon tasks (several in-game hours; typically > 100 action steps) per episode",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284077881",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276929178",
  "title": "EMMOE: A Comprehensive Benchmark for Embodied Mobile Manipulation in Open Environments",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 4,
  "publication_date": "2025-03-11",
  "months_since_pub": 18,
  "citations_per_month": 0.22,
  "artifact_name": "EMMOE",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "interpret a natural-language user instruction and execute a long-horizon everyday household task combining high-level and low-level embodied sub-tasks; re-plan after execution failures to recover progress toward the instructed goal",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "High-level task steps decompose into low-level navigation/manipulation actions, and failures in low-level execution require re-planning that can alter subsequent high-level steps, so the two levels are tightly coupled.",
  "n_goals": null,
  "tracking_demand": "Agent must track task-completion progress across high-level sub-tasks and low-level actions, plus failure/replan history, in continuous physical space across a long-horizon episode.",
  "scoring": "other:three-new-metrics-for-diverse-assessment - introduces three new metrics for more diverse assessment beyond a single pass/fail; abstract does not fully specify whether these are per-subtask checkpoint credit, but their stated purpose is more diverse (partial-credit-capable) assessment than a binary norm.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we propose Embodied Mobile Manipulation in Open Environments (EMMOE), a benchmark that requires agents to interpret user instructions and execute long-horizon everyday tasks in continuous space. EMMOE seamlessly integrates high-level and low-level embodied tasks into a unified framework, along with three new metrics for more diverse assessment.",
  "horizon_span": "EMMOE, a benchmark that requires agents to interpret user instructions and execute long-horizon everyday tasks in continuous space.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276929178",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279447732",
  "title": "Embodied Web Agents: Bridging Physical-Digital Realms for Integrated Agent Intelligence",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 19,
  "publication_date": "2025-06-18",
  "months_since_pub": 15,
  "citations_per_month": 1.27,
  "artifact_name": "Embodied Web Agents Benchmark",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "cook a recipe found via web-scale reasoning, using embodied physical actions (cooking); navigate physically using dynamic, web-sourced map data (navigation); shop by combining physical store interaction with online product/price information (shopping); plan tourism activities combining physical exploration with web knowledge (tourism); identify real-world landmarks by cross-referencing physical observation with web knowledge (geolocation)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each task requires 'coordinated reasoning across physical and digital realms', so an error or gap in the digital (web) retrieval step directly propagates into what physical actions are correct, and vice versa \u2014 the two channels are mutually constraining even though the five task types (cooking, navigation, shopping, tourism, geolocation) are otherwise largely independent of each other.",
  "n_goals": "5 task types (cooking, navigation, shopping, tourism, geolocation)",
  "tracking_demand": "Agent must maintain a consistent state across both a 3D embodied environment (physical position, observations, actions) and a web interface (retrieved facts, map data, product/price info), integrating the two continuously rather than treating them as separate phases.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports 'significant performance gaps between state-of-the-art AI systems and human capabilities' across the task suite, without describing a specific subgoal-level partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Significant performance gaps between state-of-the-art AI systems and human capabilities are found across the benchmark (no specific numeric score given in the abstract).",
  "availability": "https://embodied-web-agent.github.io/",
  "goal_span": "we construct and release the Embodied Web Agents Benchmark, which encompasses a diverse suite of tasks including cooking, navigation, shopping, tourism, and geolocation - all requiring coordinated reasoning across physical and digital realms",
  "horizon_span": "Building upon this platform, we construct and release the Embodied Web Agents Benchmark, which encompasses a diverse suite of tasks including cooking, navigation, shopping, tourism, and geolocation",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279447732",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "276317279",
  "title": "EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents",
  "year": 2025,
  "venue": "International Conference on Machine Learning",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 225,
  "publication_date": "2025-02-13",
  "months_since_pub": 19,
  "citations_per_month": 11.84,
  "artifact_name": "EmbodiedBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "complete each of 1,128 testing tasks across four environments, from high-level semantic household tasks to low-level atomic navigation/manipulation; demonstrate commonsense reasoning within tasks; understand complex instructions; exhibit spatial awareness and visual perception; engage in long-term planning",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "High-level semantic household tasks require composing multiple low-level atomic actions (navigation, manipulation), so correctly perceiving and reasoning about the scene at a low level constrains whether a high-level multi-step household goal can be achieved.",
  "n_goals": "1,128 testing tasks across four environments; six curated capability subsets",
  "tracking_demand": "The agent must track visual perception of the scene, instruction understanding, spatial layout, and long-term planning state as it composes low-level atomic actions to satisfy high-level household goals.",
  "scoring": "other:per-capability-subset-accuracy. The benchmark reports six curated subsets evaluating distinct capabilities (commonsense reasoning, complex instruction understanding, spatial awareness, visual perception, long-term planning), a per-capability (subgoal-type) breakdown in addition to overall task success (best model 28.9%).",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "The best model, GPT-4o, scores only 28.9% on average across EmbodiedBench; MLLMs excel at high-level tasks but struggle with low-level manipulation; no human/expert baseline given.",
  "availability": "https://embodiedbench.github.io",
  "goal_span": "EmbodiedBench features: (1) a diverse set of 1,128 testing tasks across four environments, ranging from high-level semantic tasks (e.g., household) to low-level tasks involving atomic actions (e.g., navigation and manipulation); and (2) six meticulously curated subsets evaluating essential agent capabilities like commonsense reasoning, complex instruction understanding, spatial awareness, visual perception, and long-term planning.",
  "horizon_span": "six meticulously curated subsets evaluating essential agent capabilities like commonsense reasoning, complex instruction understanding, spatial awareness, visual perception, and long-term planning",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276317279",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282304705",
  "title": "EmbodiedBrain: Expanding Performance Boundaries of Task Planning for Embodied Intelligence",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 3,
  "publication_date": "2025-10-23",
  "months_since_pub": 11,
  "citations_per_month": 0.27,
  "artifact_name": "EmbodiedBrain evaluation suite (General, Planning, and End-to-End Simulation Benchmarks)",
  "artifact_kind": "scenario-suite",
  "domain": "embodied-household",
  "goal_types": "complete long-horizon embodied task-planning sequences by correctly building on preceding steps (Guided Precursors); perform accurate spatial perception and adaptive execution across a novel simulation environment; satisfy General, Planning, and End-to-End Simulation benchmark criteria as three complementary evaluation axes",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Step-GRPO explicitly integrates preceding steps as 'Guided Precursors', meaning success on a later step in a long-horizon task is directly conditioned on correctly completing/using information from earlier steps in the same trajectory.",
  "n_goals": null,
  "tracking_demand": "Agent must track its own preceding action steps as guided precursors informing subsequent steps, and its progress must be verifiable across all three of the General, Planning, and End-to-End Simulation evaluation axes.",
  "scoring": "other:three-part-evaluation-general-planning-simulation - establishes a three-part evaluation system (General, Planning, End-to-End Simulation Benchmarks); abstract does not specify a formal in-task subgoal-checkpoint credit scheme, but the three-axis structure itself implies decomposed rather than single monolithic assessment.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "EmbodiedBrain achieves superior performance across all metrics, establishing a new state-of-the-art for embodied foundation models; no specific numeric score or human/expert baseline is given in the abstract.",
  "availability": "https://zterobot.github.io/EmbodiedBrain.github.io",
  "goal_span": "employs a powerful training methodology that integrates large-scale Supervised Fine-Tuning (SFT) with Step-Augumented Group Relative Policy Optimization (Step-GRPO), which boosts long-horizon task success by integrating preceding steps as Guided Precursors. ... we establish a three-part evaluation system encompassing General, Planning, and End-to-End Simulation Benchmarks, highlighted by the proposal and open-sourcing of a novel, challenging simulation environment.",
  "horizon_span": "which boosts long-horizon task success by integrating preceding steps as Guided Precursors.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "The paper's primary contribution is a foundation model (EmbodiedBrain, 7B/32B) and a training method (Step-GRPO); the benchmark/simulation-environment artifact is a secondary contribution bundled with the model release rather than a standalone, independently-focused benchmark paper.",
  "url": "https://api.semanticscholar.org/CorpusId:282304705",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279447480",
  "title": "FindingDory: A Benchmark to Evaluate Memory in Embodied Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 12,
  "publication_date": "2025-06-18",
  "months_since_pub": 15,
  "citations_per_month": 0.8,
  "artifact_name": "FindingDory",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "recall relevant historical information (images/interactions) collected potentially across multiple days; execute low-level navigation/manipulation actions based on recalled information; complete each of 60 memory-intensive embodied tasks requiring sustained engagement; scale to procedurally extended, longer/harder versions of the same tasks",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Successful task completion requires both recalling relevant past interactions and correctly acting on that recalled information via low-level navigation/manipulation, so a memory-retrieval error directly causes an action-execution failure downstream.",
  "n_goals": "60 tasks, procedurally extendable to longer/harder versions",
  "tracking_demand": "The agent must retain and retrieve relevant historical images/interactions collected across multiple days in the Habitat simulator, combining that recall with sustained contextual awareness during ongoing navigation/manipulation.",
  "scoring": "other:per-task-success-with-procedural-difficulty-scaling. The benchmark evaluates baselines integrating VLMs with low-level navigation policies on 60 tasks; the abstract highlights areas for improvement without describing a per-subgoal checkpoint scheme beyond task-level success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we introduce a new benchmark for long-range embodied tasks in the Habitat simulator. This benchmark evaluates memory-based capabilities across 60 tasks requiring sustained engagement and contextual awareness in an environment. The tasks can also be procedurally extended to longer and more challenging versions, enabling scalable evaluation of memory and reasoning.",
  "horizon_span": "their ability to incorporate long-term experience collected across multiple days and represented by vast collections of images",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279447480",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282102541",
  "title": "RoboHiMan: A Hierarchical Evaluation Paradigm for Compositional Generalization in Long-Horizon Manipulation",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 9,
  "publication_date": "2025-10-15",
  "months_since_pub": 11,
  "citations_per_month": 0.82,
  "artifact_name": "HiMan-Bench",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "complete atomic long-horizon manipulation tasks under diverse perturbations; complete compositional tasks requiring composing multiple learned manipulation skills; generalize skill composition/scheduling to perturbed or novel conditions; coordinate high-level subgoal planning with low-level execution policies",
  "goal_origin": "mixed:top-level-task-given-up-front-subgoals-generated-by-high-level-planner",
  "decomposition": "hierarchical",
  "interdependence": "High-level planner-generated subgoals constrain what the low-level policy attempts next, and under perturbations, skill composition requires policies to recover or replan, so failure at one composed-skill boundary propagates to later skill executions; the vanilla/decoupled/coupled paradigms probe this dependency directly.",
  "n_goals": null,
  "tracking_demand": "The system must track which atomic/composed skill is currently executing, how perturbations affect preconditions for subsequent skills, and whether the high-level plan needs revision given low-level execution feedback.",
  "scoring": "other:three-paradigm-diagnostic-evaluation(vanilla/decoupled/coupled). The three evaluation paradigms are explicitly designed to probe skill-composition necessity and locate planning-vs-execution bottlenecks -- a structured diagnostic rubric beyond a single binary task-success score, though no explicit numeric subgoal-checkpoint credit is stated.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://chenyt31.github.io/robo-himan.github.io/",
  "goal_span": "RoboHiMan introduces HiMan-Bench, a benchmark of atomic and compositional tasks under diverse perturbations, supported by a multi-level training dataset for analyzing progressive data scaling, and proposes three evaluation paradigms (vanilla, decoupled, coupled) that probe the necessity of skill composition and reveal bottlenecks in hierarchical architectures.",
  "horizon_span": "existing benchmarks primarily emphasize task completion in long-horizon settings, offering little insight into compositional generalization, robustness, and the interplay between planning and execution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282102541",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276703396",
  "title": "Household Task Planning with Multi-Objects State and Relationship Using Large Language Models Based Preconditions Verification",
  "year": 2025,
  "venue": "International Conference on Agents and Artificial Intelligence",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 0,
  "publication_date": null,
  "months_since_pub": 15,
  "citations_per_month": 0.0,
  "artifact_name": null,
  "artifact_kind": "dataset",
  "domain": "embodied-household",
  "goal_types": "achieve each targeted object state change (e.g. turning an appliance on/off); achieve each targeted object placement goal; verify environmental preconditions are met before executing each action; reformulate an action step automatically when a precondition is not satisfied",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each action's preconditions depend on the state left by prior actions (e.g. a container must be open before an object can be placed in it), so the verification mechanism must re-check state after every step and reformulate the plan when a precondition fails.",
  "n_goals": null,
  "tracking_demand": "The agent must track current object states, identifiers, and relationships in the environment, re-verifying preconditions before each action and updating its plan when environmental state does not match expectations.",
  "scoring": "other:success-rate-per-task-category",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "GPT-4o achieves 89.4% success on state-change tasks and 81.6% on placement tasks; ablations confirm the precondition check's significant contribution; no human baseline given.",
  "availability": null,
  "goal_span": "Our method combines simulator-derived environmental state information with an LLM-based planning to generate executable action sequences. A key feature in our system is the LLM-driven verification mechanism that assesses whether environmental preconditions are met before each action executes, automatically reformulating action steps when prerequisites are not satisfied. Experimental results using GPT-4o demonstrate strong performance, achieving 89.4% success rate on state change tasks and 81.6% on placement tasks.",
  "horizon_span": "Experimental results using GPT-4o demonstrate strong performance, achieving 89.4% success rate on state change tasks and 81.6% on placement tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276703396",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "278740205",
  "title": "LLM-BABYBENCH: Understanding and Evaluating Grounded Planning and Reasoning in LLMs",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 6,
  "publication_date": "2025-05-17",
  "months_since_pub": 16,
  "citations_per_month": 0.38,
  "artifact_name": "LLM-BabyBench",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "predict the consequences of an action on the textual BabyAI grid-world environment state (Predict task); generate a sequence of low-level actions achieving a specified objective (Plan task); decompose a high-level instruction into a coherent sequence of subgoals (Decompose task)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The Decompose task's subgoal sequence must correctly order steps such that each subgoal's preconditions are satisfied by prior subgoals, and the Plan task's low-level action sequence must correctly realize whatever subgoal/objective is specified, so correct low-level planning is contingent on correct higher-level decomposition.",
  "n_goals": null,
  "tracking_demand": "Agent must track the current grid-world state to predict action consequences, track partial plan progress when generating low-level action sequences, and track subgoal ordering/coherence when decomposing high-level instructions.",
  "scoring": "other:environment-interaction-validated-plans - provides a standardized evaluation harness including environment interaction for validating generated plans; validating plans via actual environment interaction implies step-by-step (subgoal-level) verifiability rather than a single terminal judgment.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/choukrani/llm-babybench and https://huggingface.co/datasets/salem-mbzuai/LLM-BabyBench",
  "goal_span": "this suite evaluates LLMs on three fundamental aspects of grounded intelligence: (1) predicting the consequences of actions on the environment state ($\\textbf{Predict}$ task), (2) generating sequences of low-level actions to achieve specified objectives ($\\textbf{Plan}$ task), and (3) decomposing high-level instructions into coherent subgoal sequences ($\\textbf{Decompose}$ task).",
  "horizon_span": "We detail the methodology for generating the three corresponding datasets ($\\texttt{LLM-BabyBench-Predict}$, $\\texttt{-Plan}$, $\\texttt{-Decompose}$) by extracting structured information from an expert agent operating within the text-based environment.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278740205",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279074780",
  "title": "LoHoVLA: A Unified Vision-Language-Action Model for Long-Horizon Embodied Tasks",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 32,
  "publication_date": "2025-05-31",
  "months_since_pub": 16,
  "citations_per_month": 2.0,
  "artifact_name": "LoHoSet (Ravens simulator)",
  "artifact_kind": "dataset",
  "domain": "robotics-sim",
  "goal_types": "decompose a high-level embodied goal into a sequence of sub-tasks; generate correct low-level robot actions to execute each sub-task; maintain coordination between high-level planning and low-level motion control across the whole long-horizon task",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Sub-tasks generated during high-level planning must be executed in order by low-level motion control, and errors from either level (planning mistakes or control imprecision) can derail subsequent sub-tasks in the long-horizon task.",
  "n_goals": "20 long-horizon tasks, each with 1,000 expert demonstrations",
  "tracking_demand": "The system must track which sub-tasks of the decomposed high-level goal have been completed and maintain closed-loop consistency between the high-level plan and low-level action execution across the task.",
  "scoring": "binary-final-success. The abstract reports LoHoVLA 'significantly surpasses' both hierarchical and standard VLA baselines on long-horizon tasks in the Ravens simulator; no explicit subgoal-checkpoint partial credit is described, though sub-task decomposition is part of the architecture.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we introduce LoHoSet, a dataset built on the Ravens simulator, containing 20 long-horizon tasks, each with 1,000 expert demonstrations composed of visual observations, linguistic goals, sub-tasks, and robot actions.",
  "horizon_span": "we introduce LoHoSet, a dataset built on the Ravens simulator, containing 20 long-horizon tasks, each with 1,000 expert demonstrations",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279074780",
  "provenance": "asta-find,parametric",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278910792",
  "title": "ManiTaskGen: A Comprehensive Task Generator for Benchmarking and Improving Vision-Language Agents on Embodied Decision-Making",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 4,
  "publication_date": "2025-05-27",
  "months_since_pub": 16,
  "citations_per_month": 0.25,
  "artifact_name": "ManiTaskGen",
  "artifact_kind": "scenario-suite",
  "domain": "embodied-household",
  "goal_types": "satisfy a process-based instruction requiring a specific sequence of manipulations (e.g. 'move object from X to Y'); satisfy an outcome-based abstract instruction requiring multiple manipulations to reach a goal state (e.g. 'clear the table'); generate a comprehensive, diverse, feasible set of mobile manipulation tasks for any given scene",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Outcome-based abstract instructions (e.g. clearing a table) decompose into multiple underlying object-manipulation actions whose order and feasibility depend on the scene's current object layout, so the generator must verify feasibility across the whole manipulation sequence, not just one action.",
  "n_goals": null,
  "tracking_demand": "The agent (and the task generator itself) must track which objects have been moved/placed so far relative to the target scene configuration, verifying that an outcome-based goal (e.g. a cleared table) is only satisfied once all constituent object states are achieved.",
  "scoring": "other:qualitative-validity-and-diversity-of-generated-tasks",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "The generated tasks encompass both process-based, specific instructions (e.g.,\"move object from X to Y\") and outcome-based, abstract instructions (e.g.,\"clear the table\"). We apply ManiTaskGen to both simulated and real-world scenes, demonstrating the validity and diversity of the generated tasks.",
  "horizon_span": "we introduce ManiTaskGen, a novel system that automatically generates comprehensive, diverse, feasible mobile manipulation tasks for any given scene.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278910792",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291470637",
  "title": "OceanGym: A Benchmark Environment for Underwater Embodied Agents",
  "year": 2025,
  "venue": "",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 0,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 0.0,
  "artifact_name": "OceanGym",
  "artifact_kind": "environment/simulator",
  "domain": "robotics-sim",
  "goal_types": "comprehend and fuse optical and sonar sensor data under low visibility; autonomously explore complex underwater environments; accomplish each of eight realistic underwater task domains; adapt navigation/decision-making to dynamic ocean currents",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Perception outputs (optical/sonar fusion) feed exploration and planning decisions, and dynamic currents/low visibility mean earlier navigation errors compound into later planning failures across the long-horizon objective.",
  "n_goals": "eight realistic task domains",
  "tracking_demand": "The agent must integrate perception, memory, and sequential decision-making state (explored regions, sonar/optical evidence) across a long-horizon objective under low-visibility, dynamically-changing underwater conditions.",
  "scoring": "other:comparison-to-human-expert-performance",
  "horizon_value": "0.5 hours (t_max, per decision task)",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Substantial gaps between state-of-the-art MLLM-driven agents and human experts across perception, planning, and adaptability; no specific numeric headline figure given in the abstract.",
  "availability": "https://github.com/OceanGPT/OceanGym",
  "goal_span": "OceanGym encompasses eight realistic task domains and a unified agent framework driven by Multi-modal Large Language Models (MLLMs), which integrates perception, memory, and sequential decision-making. Agents are required to comprehend optical and sonar data, autonomously explore complex environments, and accomplish long-horizon objectives under these harsh conditions.",
  "horizon_span": "the decision interval t_interval takes 30 seconds and t_max takes 0.5 hours in decision tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": "The stated 0.5-hour max is notably short for a paper framed around 'long-horizon objectives'; may reflect only one task type rather than the benchmark's overall horizon.",
  "url": "https://api.semanticscholar.org/CorpusId:291470637",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "277435033",
  "title": "REMAC: Self-Reflective and Self-Evolving Multi-Agent Collaboration for Long-Horizon Robot Manipulation",
  "year": 2025,
  "venue": "Pattern Recognition",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 11,
  "publication_date": "2025-03-28",
  "months_since_pub": 18,
  "citations_per_month": 0.61,
  "artifact_name": "REMAC multi-agent environment (on RoboCasa)",
  "artifact_kind": "environment/simulator",
  "domain": "robotics-sim",
  "goal_types": "decompose and execute long-horizon multi-robot manipulation and navigation tasks (4 task categories, 27 task styles, 50+ objects); perform pre-condition and post-condition checks in the loop to evaluate progress and refine plans; adapt plans dynamically to unexpected scene conditions (e.g. a closed microwave door) via self-evolvement; coordinate parallel task execution across multiple robots",
  "goal_origin": "mixed:given-up-front-tasks-with-self-generated-plan-adaptation",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Pre- and post-condition checks gate progression between decomposed sub-tasks, and coordinating multiple robots in parallel requires managing shared scene state and task dependencies across robots to maximize execution efficiency.",
  "n_goals": "4 task categories with 27 task styles and 50+ different objects",
  "tracking_demand": "Agent(s) must track scene state, pre/post-condition satisfaction, and per-robot task assignment across a decomposed long-horizon manipulation/navigation plan, adapting the plan when scene-specific conditions invalidate prior assumptions.",
  "scoring": "other:success-rate-and-execution-efficiency",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "REMAC boosts average success rates by 40% and execution efficiency by 52.7% over a single-robot baseline across evaluated reasoning models (DeepSeek-R1, o3-mini, QwQ, Grok3); no human baseline given.",
  "availability": null,
  "goal_span": "REMAC incorporates two key modules: a self-reflection module performing pre-condition and post-condition checks in the loop to evaluate progress and refine plans, and a self-evolvement module dynamically adapting plans based on scene-specific reasoning... boosting average success rates by 40% and execution efficiency by 52.7% over the single robot baseline.",
  "horizon_span": "we build a multi-agent environment for long-horizon robot manipulation and navigation based on RoboCasa, featuring 4 task categories with 27 task styles and 50+ different objects.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277435033",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "283556982",
  "title": "ResponsibleRobotBench: Benchmarking Responsible Robot Manipulation using Multi-modal Large Language Models",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 3,
  "publication_date": "2025-12-03",
  "months_since_pub": 9,
  "citations_per_month": 0.33,
  "artifact_name": "ResponsibleRobotBench",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "detect and mitigate risks (electrical, chemical, human-related hazards) during a multi-stage manipulation task; reason about safety and physically grounded planning across the task; plan and execute sequences of manipulation actions; engage human assistance when necessary rather than proceeding unsafely",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each of the 23 multi-stage tasks requires risk detection at one stage to inform safe planning at later stages, and a failure to mitigate an earlier-stage hazard can make later manipulation stages unsafe or infeasible; the decision to engage human assistance is itself conditioned on the risk assessed so far.",
  "n_goals": "23 multi-stage tasks spanning diverse risk types and varying levels of physical/planning complexity",
  "tracking_demand": "Agent must track detected hazards (electrical, chemical, human-related), current safety status, and multi-stage plan progress, deciding when to escalate to human assistance across the task.",
  "scoring": "other:success-safety-safe-success-rate - includes standardized metrics such as success rate, safety rate, and safe success rate (task success achieved without unsafe incidents); an explicit multi-metric scoring scheme beyond simple task completion, though not framed as fine-grained per-stage partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://sites.google.com/view/responsible-robotbench",
  "goal_span": "This benchmark consists of 23 multi-stage tasks spanning diverse risk types, including electrical, chemical, and human-related hazards, and varying levels of physical and planning complexity. These tasks require agents to detect and mitigate risks, reason about safety, plan sequences of actions, and engage human assistance when necessary.",
  "horizon_span": "This benchmark consists of 23 multi-stage tasks spanning diverse risk types, including electrical, chemical, and human-related hazards, and varying levels of physical and planning complexity.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283556982",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279250267",
  "title": "RoboCerebra: A Large-scale Benchmark for Long-horizon Robotic Manipulation Evaluation",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 33,
  "publication_date": "2025-06-07",
  "months_since_pub": 15,
  "citations_per_month": 2.2,
  "artifact_name": "RoboCerebra",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "decompose a high-level household instruction into a sequence of dependent subtasks (via GPT-generated instructions); correctly execute each subtask in the sequence despite dynamic object variations; sustain planning, reflection, and memory (System 2 reasoning) across the full extended action sequence, not just react (System 1) to the current observation",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later subtasks in the sequence depend on the state left by earlier subtasks (e.g. object positions after dynamic variations), and the hierarchical framework's high-level VLM planner must correctly track this evolving state to issue valid next-subtask plans to the low-level VLA controller.",
  "n_goals": "1,000 human-annotated trajectories across 100 task variants, with dynamic object variations, decomposed by GPT into subtask sequences",
  "tracking_demand": "The high-level planner must track evolving object/environment state across an average trajectory length of 2,972.4 simulation steps (about 6x longer than existing long-horizon manipulation datasets), correctly sequencing and reflecting on subtasks throughout.",
  "scoring": "milestone-rubric -- an evaluation protocol 'targeting planning, reflection, and memory through structured System 1-System 2 interaction' implies scoring across distinct cognitive dimensions/checkpoints rather than a single end-of-trajectory binary success, though the abstract does not give an explicit numeric rubric.",
  "horizon_value": "average trajectory length reaches 2,972.4 simulation steps, about 6x longer than existing long-horizon manipulation datasets",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/trajectory",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "RoboCerebra includes: (1) a large-scale simulation dataset with extended task horizons and diverse subtask sequences in household environments; (2) a hierarchical framework combining a high-level VLM planner with a low-level vision-language-action (VLA) controller; and (3) an evaluation protocol targeting planning, reflection, and memory through structured System 1-System 2 interaction.",
  "horizon_span": "the average trajectory length reaches 2,972.4 simulation steps ... about 6\u00d7 longer than existing long-horizon manipulation datasets",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279250267",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281706645",
  "title": "RoboPilot: Generalizable Dynamic Robotic Manipulation with Dual-thinking Modes",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "robotics-sim",
  "citation_count": 1,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 0.08,
  "artifact_name": "RoboPilot-Bench",
  "artifact_kind": "benchmark",
  "domain": "robotics-sim",
  "goal_types": "execute complex or long-horizon robotic manipulation tasks despite environmental changes; recognize infeasible tasks rather than attempting them blindly; recover from execution errors via closed-loop replanning; succeed across each of 21 tasks spanning 10 manipulation categories",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Within a task, primitive actions are chained via structured task planning, and feedback triggers replanning when execution errors or dynamic environmental changes occur, so later actions must be revised based on the success/failure of earlier ones rather than executed open-loop.",
  "n_goals": "21 tasks across 10 categories",
  "tracking_demand": "The agent must track execution feedback and environmental state to detect deviations or errors requiring replanning, and separately recognize when a task is infeasible altogether, across a long-horizon sequence of primitive manipulation actions.",
  "scoring": "other:task-success-rate-relative-to-baselines-plus-infeasible-task-and-failure-recovery-subtests",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "RoboPilot outperforms state-of-the-art baselines by 25.9% in task success rate on RoboPilot-Bench, with real-world deployment on an industrial robot further validating robustness; no explicit human-operator baseline is reported.",
  "availability": null,
  "goal_span": "we introduce RoboPilot-Bench, a benchmark spanning 21 tasks across 10 categories, including infeasible-task recognition and failure recovery. Experiments show that RoboPilot outperforms state-of-the-art baselines by 25.9% in task success rate",
  "horizon_span": "Despite rapid progress in autonomous robotics, executing complex or long-horizon tasks remains a fundamental challenge.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281706645",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276249312",
  "title": "Robotouille: An Asynchronous Planning Benchmark for LLM Agents",
  "year": 2025,
  "venue": "International Conference on Learning Representations",
  "tier": "in",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 38,
  "publication_date": "2025-02-06",
  "months_since_pub": 19,
  "citations_per_month": 2.0,
  "artifact_name": "Robotouille",
  "artifact_kind": "environment/simulator",
  "domain": "embodied-household",
  "goal_types": "complete overlapping cooking sub-tasks that must be scheduled around each other (e.g. one dish while another cooks); handle interruptions to an in-progress plan without dropping earlier commitments; reason over states/actions that must occur in parallel versus strictly sequentially",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Kitchen tasks overlap in time and share resources (stations, ingredients, time delays), so asynchronous scheduling requires tracking which sub-tasks can proceed in parallel and which must wait, and handling interruptions without abandoning earlier commitments.",
  "n_goals": null,
  "tracking_demand": "Agent must track the state and expected completion time of multiple concurrently in-progress cooking actions, remember interrupted sub-goals, and self-audit its plan as time delays resolve.",
  "scoring": "binary-final-success \u2014 the abstract reports overall task success rates (47% synchronous vs. 11% asynchronous for ReAct/gpt-4o) with no mention of partial/subgoal-level credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "ReAct (gpt-4o) achieves 47% on synchronous tasks but only 11% on asynchronous tasks; no human baseline given.",
  "availability": "https://github.com/portal-cornell/robotouille",
  "goal_span": "We introduce Robotouille, a challenging benchmark environment designed to test LLM agents' ability to handle long-horizon asynchronous scenarios.",
  "horizon_span": "current benchmarks focus primarily on short-horizon tasks and do not evaluate such asynchronous planning capabilities",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276249312",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281421862",
  "title": "Evaluating Multimodal Large Language Models with Daily Composite Tasks in Home Environments",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "embodied-household",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "embodied-household",
  "citation_count": 1,
  "publication_date": "2025-09-22",
  "months_since_pub": 12,
  "citations_per_month": 0.08,
  "artifact_name": "unnamed Daily Composite Tasks benchmark",
  "artifact_kind": "benchmark",
  "domain": "embodied-household",
  "goal_types": "correctly perform object-understanding sub-tasks (e.g. counting/categorizing objects) within a composite task; correctly perform spatial-intelligence sub-tasks within the same composite task; correctly perform social-activity sub-tasks within the same composite task, all within one dynamic simulated home environment",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "The abstract frames each composite task as requiring a wide range of capabilities spanning three domains jointly, but does not specify explicit ordering, shared-resource, or mutual-exclusion constraints between the three sub-domains within one task; none stated.",
  "n_goals": "eight types of embodied composite tasks spanning three core domains (object understanding, spatial intelligence, social activity), evaluated on 17 leading MLLMs",
  "tracking_demand": "The agent must track and integrate information across the three jointly-required capability domains (object understanding, spatial intelligence, social activity) within one dynamic, simulated home environment to complete a composite task.",
  "scoring": "binary-final-success -- performance is reported as consistently poor across all three domains for all 17 evaluated models; the abstract does not describe an explicit subgoal-checkpoint partial-credit scheme distinguishing performance within a single composite task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "In the current work, we designed a set of composite tasks inspired by common daily activities observed in early childhood development. Within a dynamic and simulated home environment, these tasks span three core domains: object understanding, spatial intelligence, and social activity.",
  "horizon_span": "we designed a set of composite tasks inspired by common daily activities observed in early childhood development ... We introduced eight types of embodied composite tasks.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281421862",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287843434",
  "title": "AgentClinic: a multimodal benchmark for tool-using clinical AI agents",
  "year": 2026,
  "venue": "npj Digital Medicine",
  "tier": "relevant",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:clinical-diagnostic-simulation",
  "citation_count": 21,
  "publication_date": "2026-04-27",
  "months_since_pub": 5,
  "citations_per_month": 4.2,
  "artifact_name": "AgentClinic",
  "artifact_kind": "benchmark",
  "domain": "other:clinical-diagnostic-simulation",
  "goal_types": "engage in sequential clinical decision-making across a diagnostic encounter (patient interaction, exam/imaging requests); collect multimodal data under incomplete information via various tools (e.g., notebook, retrieval); reach a correct diagnosis across nine medical specialties and seven languages",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Diagnostic decisions build on previously gathered patient information and test results, so the order of information-gathering actions affects what evidence is available for later differential-diagnosis steps; diagnostic accuracy drops sharply when done sequentially versus as static QA.",
  "n_goals": null,
  "tracking_demand": "Agent must track patient information collected so far (multimodal exam/imaging results), notes taken across a case (persisting via a notebook tool), and its evolving working diagnosis across the encounter.",
  "scoring": "other:diagnostic-accuracy-drop-vs-static-qa - reports diagnostic accuracy that can drop to below a tenth of the static MedQA accuracy when done sequentially; abstract does not describe an explicit per-step subgoal-checkpoint credit scheme beyond final diagnostic accuracy, though patient-centric metrics are also explored.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Diagnostic accuracy in AgentClinic's sequential format can drop to below a tenth of the static-QA accuracy; agents built on Claude-3.5 outperform other backbones in most settings; Llama-3 shows up to 92% relative improvement with a persistent notebook tool; no explicit human/expert clinician baseline is given beyond a clinical reader study.",
  "availability": null,
  "goal_span": "we introduce AgentClinic, a multimodal agent benchmark for evaluating LLMs in simulated clinical environments that include patient interactions, multimodal data collection under incomplete information, and the usage of various tools, resulting in an in-depth evaluation across nine medical specialties and seven languages.",
  "horizon_span": "We find that solving MedQA problems in the sequential decision-making format of AgentClinic is considerably more challenging, resulting in diagnostic accuracies that can drop to below a tenth of the original accuracy.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287843434",
  "provenance": "asta-find,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288860800",
  "title": "ClinEnv: An Interactive Multi-Stage Long Horizon EHR Environment for Agents",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:clinical-decision-making",
  "citation_count": 2,
  "publication_date": "2026-06-01",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "ClinEnv",
  "artifact_kind": "benchmark",
  "domain": "other:clinical-decision-making",
  "goal_types": "progress through an ordered, per-case sequence of clinical decision stages; actively query four specialized information agents before committing to a decision at each stage; commit to correct medications, procedures, and diagnoses at each stage; avoid redundant information-gathering queries as the case progresses",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Decisions are sequential and irreversible: committing to a decision at one stage forecloses reconsideration, and difficulty concentrates in later stages, so the quality of earlier information-gathering conditions later, more consequential decisions (e.g., management actions vs. discharge diagnosis).",
  "n_goals": null,
  "tracking_demand": "Agent must actively query four specialized information agents at every decision stage before committing to medications, procedures, and diagnoses, tracking what has already been queried to avoid redundant queries as the case progresses.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Across seven models, the strongest reaches only 0.31 overall decision F1, with discharge-diagnosis recovery (0.51 F1) far exceeding management-action recovery (0.17 F1); no human/attending-physician baseline is numerically given.",
  "availability": null,
  "goal_span": "Each case is automatically constructed into an ordered sequence of decision stages; at every stage the model must actively query four specialized agents before committing to medications, procedures, and diagnoses. ClinEnv scores both what the model decides, through deterministic ontology-grounded matching, and how it gathers information.",
  "horizon_span": "Each case is automatically constructed into an ordered sequence of decision stages; at every stage the model must actively query four specialized agents before committing to medications, procedures, and diagnoses.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288860800",
  "provenance": "web-registry",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288258628",
  "title": "CodeClinic: Evaluating Automation of Coding Skills for Clinical Reasoning Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:clinical-diagnostic-simulation",
  "citation_count": 0,
  "publication_date": "2026-05-10",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "CodeClinic",
  "artifact_kind": "benchmark",
  "domain": "other:clinical-diagnostic-simulation",
  "goal_types": "in the longitudinal ICU-surveillance task, make a structured monitoring decision every four hours across 25 findings and eight clinical families for the duration of a patient trajectory; in the compositional information-seeking task, answer queries whose difficulty is stratified by compositional dependency depth across 259 tasks in 9 domains; synthesize and compose reusable clinical skills rather than relying on a fixed toolbox",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "In the longitudinal setting, each 4-hourly decision follows from the patient trajectory established by prior decisions/observations; in the compositional setting, higher compositional-dependency-depth tasks require correctly chaining more component clinical skills, so failures at lower depths propagate to failures at higher depths.",
  "n_goals": "longitudinal: structured decisions every 4 hours across 25 findings/8 clinical families per patient trajectory; compositional: 63k instances across 259 tasks in 9 domains, stratified by dependency depth",
  "tracking_demand": "Agent must track the patient's evolving clinical trajectory and make a fresh structured decision every four hours in the longitudinal task, and/or track which reusable clinical skills it has synthesized/composed so far in the compositional task.",
  "scoring": "other:accuracy-plus-token-efficiency-across-two-tasks - measures consistency (accuracy) improvements and per-query token-usage reduction (up to 40%) versus zero-shot code generation across the two complementary tasks; abstract does not describe an explicit per-decision or per-depth-level subgoal checkpoint credit scheme beyond overall consistency.",
  "horizon_value": "every 4 hours (decision cadence within a longitudinal ICU patient trajectory; total trajectory length not stated)",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per decision interval within the longitudinal episode; the total episode length in hours/decisions is not stated",
  "horizon_stated": "yes",
  "headline_result": "The offline autoformalization pipeline's resulting skill libraries improve consistency while reducing per-query token usage by up to 40% compared with zero-shot code generation; no explicit human/expert clinician baseline is reported.",
  "availability": null,
  "goal_span": "The benchmark contains two complementary tasks: longitudinal ICU surveillance and compositional information seeking. The longitudinal setting simulates monitoring patient trajectories with structured decisions every four hours across 25 findings and eight clinical families, while the compositional setting spans 63k instances across 259 tasks in nine domains and is stratified by compositional dependency depth to evaluate increasingly complex multi-step reasoning.",
  "horizon_span": "The longitudinal setting simulates monitoring patient trajectories with structured decisions every four hours across 25 findings and eight clinical families.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288258628",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289629996",
  "title": "Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "other:clinical-diagnostic-simulation",
  "citation_count": 0,
  "publication_date": "2026-06-24",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "Healink",
  "artifact_kind": "benchmark",
  "domain": "other:clinical-diagnostic-simulation",
  "goal_types": "maintain continuity of care across longitudinal, patient-specific post-discharge follow-up interactions; generate prescription-grounded, traceable responses reflecting the correct patient/phenotypic/intervention history; prevent cross-departmental drug conflicts while integrating fragmented histories across clinical departments",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Historical clinical records and prior prescriptions across departments constrain what new prescriptions or advice are safe, so later follow-up responses must be consistent with the full longitudinal, cross-departmental history rather than any single encounter in isolation.",
  "n_goals": "400 continuous and 85 highly complex real-world follow-up cases, plus the webMedQA benchmark",
  "tracking_demand": "System must track a patient's full longitudinal clinical history (vectorized records, phenotypic/intervention dimensions) across departments and follow-up interactions, actively cross-checking for drug conflicts as new prescriptions are considered.",
  "scoring": "LLM-judge-rubric - a rigorous single-blind evaluation by clinical experts scored authoritativeness and clinical safety, with the framework outperforming human physician baselines on both dimensions per the abstract; abstract does not describe a formal automated per-turn subgoal checkpoint beyond this expert scoring.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "In a single-blind evaluation by clinical experts, Healink outperformed human physician baselines in both authoritativeness and clinical safety on 400 continuous and 85 highly complex follow-up cases plus the webMedQA benchmark; specific numeric scores are not given in the abstract.",
  "availability": null,
  "goal_span": "We evaluated Healink on a dataset comprising 400 continuous and 85 highly complex real-world follow-up cases, alongside the webMedQA benchmark. In a rigorous single-blind evaluation conducted by clinical experts, the framework outperformed human physician baselines in both authoritativeness and clinical safety.",
  "horizon_span": "We evaluated Healink on a dataset comprising 400 continuous and 85 highly complex real-world follow-up cases, alongside the webMedQA benchmark.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289629996",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289690338",
  "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:healthcare-agentic-workflows",
  "citation_count": 7,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 2.33,
  "artifact_name": "HealthAgentBench",
  "artifact_kind": "benchmark",
  "domain": "other:healthcare-agentic-workflows",
  "goal_types": "explore raw, heterogeneous healthcare data under minimal instructions; operate within a complex clinical environment to execute a multi-step, end-to-end task; develop research modeling pipelines over EHR data; complete tasks spanning 7 categories across the patient journey and multiple modalities",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Each task replicates an end-to-end clinical workflow, so producing the final clinical output depends on correctly exploring and using the raw healthcare data retrieved in earlier steps; the 7 task categories are otherwise largely independent environments.",
  "n_goals": "54 tasks across 7 categories",
  "tracking_demand": "The agent must track what it has already discovered while exploring raw healthcare data and its intermediate modeling/analysis steps, so its final multi-step solution reflects the full workflow rather than a shortcut response.",
  "scoring": "binary-final-success. The abstract reports a single overall task success rate per agent (~42% for the best agent) as 'a single, interpretable metric,' with no subgoal-level partial credit described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Codex GPT-5.5 is the strongest and most cost-effective agent at ~42% task success rate on HealthAgentBench; no human/expert clinician baseline is reported in the abstract.",
  "availability": "https://github.com/microsoft/HealthAgentBench",
  "goal_span": "Each task is designed to replicate an end-to-end clinical workflow: given minimal instructions, an agent must explore raw healthcare data, operate within a complex environment, and execute multi-step solutions that go beyond naive prompting.",
  "horizon_span": "We introduce HealthAgentBench, a suite of 54 agentic healthcare tasks across 7 categories each with its unique environment.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289690338",
  "provenance": "web-registry",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285240965",
  "title": "MedMCP-Calc: Benchmarking LLMs for Realistic Medical Calculator Scenarios via MCP Integration",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": true,
  "domain_detail": "other:clinical-medical-calculator-workflow",
  "citation_count": 3,
  "publication_date": "2026-01-30",
  "months_since_pub": 8,
  "citations_per_month": 0.38,
  "artifact_name": "MedMCP-Calc",
  "artifact_kind": "benchmark",
  "domain": "other:clinical-medical-calculator-workflow",
  "goal_types": "proactively acquire relevant patient data from an EHR database; select the scenario-appropriate medical calculator among alternatives; perform multi-step computation using retrieved data and the selected calculator; retrieve external reference information when needed; complete each of 118 scenario tasks across 4 clinical domains",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Correct computation depends on first correctly acquiring the right EHR data and selecting the appropriate calculator for the given fuzzy scenario, so an error early in the chain (wrong data pulled, wrong calculator chosen) propagates to the final numeric result.",
  "n_goals": "118 scenario tasks across 4 clinical domains",
  "tracking_demand": "The agent must track which EHR fields it has retrieved via iterative SQL-based database interaction, which calculator it has selected, and intermediate computed values, since evaluation is process-level rather than only checking the final number.",
  "scoring": "other:process-level-evaluation-across-workflow-stages",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even top performers like Claude Opus 4.5 show substantial gaps (calculator selection under fuzzy queries, poor iterative SQL interaction, reluctance to use external tools); the authors' own CalcMate model achieves state-of-the-art among open-source models, but no explicit human/expert clinician score is given.",
  "availability": "https://github.com/SPIRAL-MED/MedMCP-Calc",
  "goal_span": "MedMCP-Calc comprises 118 scenario tasks across 4 clinical domains, featuring fuzzy task descriptions mimicking natural queries, structured EHR database interaction, external reference retrieval, and process-level evaluation.",
  "horizon_span": "requiring proactive EHR data acquisition, scenario-dependent calculator selection, and multi-step computation",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285240965",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288260220",
  "title": "MedMemoryBench: Benchmarking Agent Memory in Personalized Healthcare",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "other:healthcare-personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-05-12",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "MedMemoryBench",
  "artifact_kind": "benchmark",
  "domain": "other:healthcare-personal-assistant-memory",
  "goal_types": "accumulate and correctly retain/retrieve clinically relevant patient information across many sessions; sustain retrieval and reasoning robustness despite memory saturation from continued information influx; correctly evaluate an agent's memory 'live', as it is constructed, rather than only after the fact (evaluate-while-constructing streaming protocol)",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later sessions' correct handling depends on the clinical state accumulated from earlier sessions, and the paper explicitly identifies 'memory saturation' -- where sustained information influx degrades retrieval/reasoning -- as a critical phenomenon linking accumulated history to later performance.",
  "n_goals": "approximately 2,000 sessions and 16,000 interaction turns across clinically grounded synthetic patient archetypes",
  "tracking_demand": "The agent's memory system must retain and correctly retrieve clinically relevant information across roughly 2,000 sessions and 16,000 interaction turns per patient archetype, resisting degradation from memory saturation and noise while supporting complex medical reasoning.",
  "scoring": "other:streaming-evaluate-while-constructing -- MedMemoryBench pioneers an 'evaluate-while-constructing' streaming assessment protocol that evaluates memory continuously as it accumulates, rather than only via a single final score, giving process-level rather than purely outcome-level credit.",
  "horizon_value": "approximately 2,000 sessions and 16,000 interaction turns",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "dataset-wide total across clinically grounded synthetic patient archetypes (not a single fixed per-patient session count)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "This process yields a massive, expertly validated dataset comprising approximately 2,000 sessions and 16,000 interaction turns. Crucially, MedMemoryBench departs from traditional static evaluations by pioneering an \"evaluate-while-constructing\" streaming assessment protocol, which precisely mirrors dynamic memory accumulation in production environments. Furthermore, we formalize and systematically investigate the critical phenomenon of memory saturation, where sustained information influx actively degrades retrieval and reasoning robustness.",
  "horizon_span": "This process yields a massive, expertly validated dataset comprising approximately 2,000 sessions and 16,000 interaction turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288260220",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "sessions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277435004",
  "title": "Self-Evolving Multi-Agent Simulations for Realistic Clinical Interactions",
  "year": 2025,
  "venue": "International Conference on Medical Image Computing and Computer-Assisted Intervention",
  "tier": "relevant",
  "family": "healthcare-clinical",
  "family_source": "extracted-domain-other",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "other:clinical-diagnostic-simulation",
  "citation_count": 48,
  "publication_date": "2025-03-28",
  "months_since_pub": 18,
  "citations_per_month": 2.67,
  "artifact_name": "MedAgentSim",
  "artifact_kind": "environment/simulator",
  "domain": "other:clinical-diagnostic-simulation",
  "goal_types": "as doctor agent, request relevant medical examinations and imaging results from a measurement agent across multi-turn conversations to reach a diagnosis; iteratively refine diagnostic strategy across successive patient interactions via self-improvement mechanisms",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each examination/imaging result requested from the measurement agent informs which further tests or questions the doctor agent pursues next, and self-improvement mechanisms carry learned strategy forward across patients.",
  "n_goals": null,
  "tracking_demand": "Doctor agent must track which exams/imaging results it has already requested and received, its evolving working diagnosis, and experience-based knowledge accumulated as it interacts with more patients over time.",
  "scoring": "other:diagnostic-interaction-evaluation-benchmark - introduces 'an evaluation benchmark for assessing the LLM's ability to engage in dynamic, context-aware diagnostic interactions'; abstract does not specify whether scoring is per-turn/subgoal or purely final-diagnosis accuracy.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://medagentsim.netlify.app/",
  "goal_span": "Unlike prior approaches, our framework requires doctor agents to actively engage with patients through multi-turn conversations, requesting relevant medical examinations (e.g., temperature, blood pressure, ECG) and imaging results (e.g., MRI, X-ray) from a measurement agent to mimic the real-world diagnostic process.",
  "horizon_span": "our framework requires doctor agents to actively engage with patients through multi-turn conversations, requesting relevant medical examinations (e.g., temperature, blood pressure, ECG) and imaging results (e.g., MRI, X-ray) from a measurement agent to mimic the real-world diagnostic process.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277435004",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289187200",
  "title": "DailyReport: An Open-ended Benchmark for Evaluating Search Agents on Daily Search Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "other:deep-research-search-agents",
  "citation_count": 0,
  "publication_date": "2026-06-11",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "DailyReport",
  "artifact_kind": "benchmark",
  "domain": "other:deep-research-search-agents",
  "goal_types": "autonomously explore web sources and synthesize information into a comprehensive response to an open-ended daily search query; satisfy each of a task's associated cascade rubrics across disentangled evaluation dimensions",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each task is explicitly decomposed into subtasks evaluated with 'cascade rubrics across disentangled dimensions', implying dimension-level scores build on (cascade from) subtask-level judgments, and a 'cascade performance attribution' method is used to trace how sub-level performance affects the aggregated dimension/user-preference scores.",
  "n_goals": "150 tasks with 3,546 associated rubrics (~23.6 rubrics/task on average)",
  "tracking_demand": "Search agent must track which subtasks/rubric dimensions of an open-ended query it has satisfied, aggregating cascade performance across disentangled dimensions into an interpretable, user-centric final score.",
  "scoring": "milestone-rubric \u2014 'Each task is decomposed into subtasks and evaluated with cascade rubrics across disentangled dimensions', giving explicit sub-task/rubric-level (subgoal) credit rather than a single binary outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Results on 17 agentic systems show 'current systems still fall short of users' expectations' (no single numeric headline score or human baseline given).",
  "availability": "https://github.com/AGI-Eval-Official/DailyReport",
  "goal_span": "Each task is decomposed into subtasks and evaluated with cascade rubrics across disentangled dimensions.",
  "horizon_span": "It contains 150 open-ended tasks with 3,546 associated rubrics, capturing widely discussed and timely information demands of real-world users",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289187200",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283897826",
  "title": "DeepSearchQA: Bridging the Comprehensiveness Gap for Deep Research Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "tool-use-API",
  "info_seeking_component": true,
  "domain_detail": "tool-use-API",
  "citation_count": 39,
  "publication_date": "2026-01-28",
  "months_since_pub": 8,
  "citations_per_month": 4.88,
  "artifact_name": "DeepSearchQA",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "systematically collate fragmented information from disparate open-web sources; de-duplicate and resolve entities to ensure precision in the exhaustive answer list; reason about stopping criteria within an open-ended search space; complete each causal-chain step, where later steps depend on successful completion of the previous one",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each task is structured as a causal chain where discovering information for one step is dependent on the successful completion of the previous one, so an error or omission in an earlier retrieval step directly blocks or corrupts every subsequent step.",
  "n_goals": "900 prompts across 17 fields",
  "tracking_demand": "The agent must track which sources it has already collated, de-duplicate overlapping entities, maintain the causal-chain dependency state (what has been successfully resolved so far), and decide when to stop searching within an open-ended web space.",
  "scoring": "other:recall-precision-balance-against-verifiable-answer-sets",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Even the most advanced models struggle to balance high recall with precision, with failure modes ranging from premature stopping to hedging with low-confidence answers; no single numeric headline figure given.",
  "availability": null,
  "goal_span": "Each task is structured as a causal chain, where discovering information for one step is dependent on the successful completion of the previous one, stressing long-horizon planning and context retention. All tasks are grounded in the open web with objectively verifiable answer sets.",
  "horizon_span": "We introduce DeepSearchQA, a 900-prompt benchmark for evaluating agents on difficult multi-step information-seeking tasks across 17 different fields.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283897826",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291328748",
  "title": "EarthVerse: Benchmarking Scientific Agents Across Dynamic Earth Systems and Natural Hazards",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "scientific-discovery",
  "citation_count": 0,
  "publication_date": "2026-08-24",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "EarthVerse",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "inspect heterogeneous event packages and choose compatible evidence for a natural-hazard investigation; execute transparent calculations reconciling differences across sources; preserve provenance across evidence, scales, units, and calculations in the final answer; produce each of the fine-grained answer units defined by the task's executable ground truth",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Correctly reconciling source differences and preserving provenance at the end depends on having selected compatible evidence and executed correct calculations earlier in the investigation, so an error in one fine-grained answer unit propagates downstream.",
  "n_goals": "405 reproducible tasks grounded in 199 documented events and 19 hazard families",
  "tracking_demand": "Agent must track evidence provenance, scales, units, and intermediate calculation results across a multi-stage investigation, since fine-grained answer units are each checked against executable ground truth.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best mean answer-unit accuracy across 25 evaluated model/agent systems is 84.65%, but the highest Strict@95 (fully correct chains) is only 34.81%; no human-expert baseline given.",
  "availability": null,
  "goal_span": "We provide executable ground truth that decomposes each task into fine-grained answer units, together with task-specific rubrics that assess the supporting research process while allowing multiple valid paths... the best mean answer-unit accuracy is 84.65%, while the highest Strict@95 is only 34.81%.",
  "horizon_span": "Its 405 reproducible tasks are grounded in 199 documented events and 19 hazard families.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291328748",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291832235",
  "title": "Evaluating Deep-Search Agents under Hierarchical Web Evidence Poisoning",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "other:deep-research-evidence-verification",
  "citation_count": 0,
  "publication_date": "2026-09-05",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "HAE-GEO",
  "artifact_kind": "benchmark",
  "domain": "other:deep-research-evidence-verification",
  "goal_types": "search and retrieve web evidence for a consumer-decision query while a poisoning attack (of increasing sophistication, L1-L3) is present; recognize/verify suspicious evidence rather than adopting it uncritically; revise any already-adopted poisoned claims and recover to a trustworthy final recommendation",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Whether the agent 'recovers' by the final recommendation depends on whether it verified and revised evidence adopted earlier in the trajectory, so early uncritical adoption of poisoned content directly constrains whether recovery is still possible later.",
  "n_goals": "72,039 clean pages and 770 poisoned pages per attack level, spanning 8 product categories and 154 brands, across 3 escalating attack levels (L1-L3)",
  "tracking_demand": "The agent must track which evidence it has already adopted (and whether that evidence was verified or poisoned), across a multi-turn Search-Scrape interaction, and must be able to revise earlier adopted claims before finalizing its recommendation.",
  "scoring": "other:full-trajectory-behavioral-plus-rubric -- evaluation combines deterministic behavioral measures with six semantic rubric dimensions tracking the full exposure-to-recovery trajectory, not a single binary final-answer score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/zhonganbi102-netizen/HAE-GEO",
  "goal_span": "We introduce HAE-GEO, a benchmark that tracks the full trajectory from exposure to recovery under progressively more persuasive Web poisoning. Agents interact via a multi-turn Search-Scrape interface across three attack levels (L1 direct assertion, L2 contextual camouflage, and L3 apparent corroboration)... Evaluation combines deterministic behavioral measures with six semantic rubric dimensions.",
  "horizon_span": "Agents interact via a multi-turn Search-Scrape interface across three attack levels (L1 direct assertion, L2 contextual camouflage, and L3 apparent corroboration), supported by a controlled corpus of 72,039 clean pages and 770 poisoned pages per level spanning 8 product categories and 154 brands.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291832235",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287635317",
  "title": "MedProbeBench: Systematic Benchmarking at Deep Evidence Integration for Expert-level Medical Guideline",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-04-20",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "MedProbeBench (MedProbe-Eval)",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "retrieve, synthesize, and reason over large-scale external medical evidence to reach expert-level judgment; satisfy 1,200+ task-adaptive rubric criteria for one produced clinical guideline; ensure each of 5,130+ atomic claims in the guideline is precisely evidenced (fine-grained evidence verification)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Producing a single valid clinical guideline requires jointly satisfying many independently-checkable rubric criteria and atomic-claim verifications, so a failure on any one atomic claim or rubric item degrades overall guideline quality without necessarily blocking the others (each is separately verifiable).",
  "n_goals": "1,200+ task-adaptive rubric criteria; 5,130+ atomic claims, per produced guideline",
  "tracking_demand": "System must track which rubric criteria have been satisfied and which atomic claims have been verified against source evidence as it synthesizes a guideline from large-scale external knowledge.",
  "scoring": "milestone-rubric \u2014 'Holistic Rubrics with 1,200+ task-adaptive rubric criteria for comprehensive quality assessment' plus 'Fine-grained Evidence Verification...grounded in 5,130+ atomic claims', both explicit forms of granular, criterion-level (subgoal-level) credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Evaluation of 17 LLMs and deep research agents 'reveals critical gaps in evidence integration and guideline generation' relative to expert-level guideline development (no single numeric headline score given).",
  "availability": "https://github.com/uni-medical/MedProbeBench",
  "goal_span": "(1) Holistic Rubrics with 1,200+ task-adaptive rubric criteria for comprehensive quality assessment, and (2) Fine-grained Evidence Verification for rigorous validation of evidence precision, grounded in 5,130+ atomic claims",
  "horizon_span": "we introduce MedProbeBench, the first benchmark leveraging high-quality clinical guidelines as expert-level references",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287635317",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291902282",
  "title": "Mr.LHDR: A Benchmark for Multimodal Real-World Long-Horizon Deep Research Agents",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "other:deep-research",
  "citation_count": 0,
  "publication_date": "2026-09-10",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Mr.LHDR",
  "artifact_kind": "benchmark",
  "domain": "other:deep-research",
  "goal_types": "derive an average of 12.1 necessary intermediate conclusions along a hidden Node-Relation dependency graph before reaching the final answer; integrate multimodal evidence (images, maps, PDFs, logos, charts, tables, video frames) where at least one non-text element changes the reasoning state; maintain correctness of intermediate conclusions consistent with annotated dependencies, not just the final answer",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each question is built from a hidden Node-Relation graph with a mean dependency depth of 10.4, so later intermediate conclusions strictly depend on correctly deriving earlier ones in the chain; an error anywhere in the chain breaks all downstream conclusions.",
  "n_goals": "average 12.1 necessary intermediate conclusions per question; mean dependency depth 10.4; eight evidence categories",
  "tracking_demand": "Agent must track which intermediate conclusions it has established so far, their dependency relationships, and integrate incremental multimodal evidence that can change the reasoning state, across long, irreducible evidence chains.",
  "scoring": "subgoal-checkpoint-partial-credit - explicitly evaluates both final answers and the correctness of intermediate conclusions under annotated dependencies, using Overall Accuracy (OA), Strict Accuracy (SA), Checklist Score (CS), and a Dependency-Aware Checklist Score (DACS); an explicit, fine-grained subgoal-level credit scheme.",
  "horizon_value": "avg 12.1 necessary intermediate conclusions per question (mean dependency depth 10.4)",
  "horizon_unit": "other:intermediate-conclusions-per-question",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/question - each question requires traversing this many chained intermediate conclusions before a final answer",
  "horizon_stated": "yes",
  "headline_result": "The strongest system achieves only 43.1% Overall Accuracy and 34.3% Strict Accuracy; removing images reduces DACS by 12.6 points; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Each question is constructed from a hidden Node-Relation graph and requires an average of 12.1 necessary intermediate conclusions with a mean dependency depth of 10.4 before reaching a short, unique, and verifiable answer.",
  "horizon_span": "Each question is constructed from a hidden Node-Relation graph and requires an average of 12.1 necessary intermediate conclusions with a mean dependency depth of 10.4 before reaching a short, unique, and verifiable answer.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291902282",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289032019",
  "title": "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "scientific-discovery",
  "citation_count": 8,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 2.0,
  "artifact_name": "ResearchClawBench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "re-discover a target published paper's scientific artifacts (methods, findings) using only related literature and raw data, with the target paper hidden; satisfy each of several expert-curated, weighted multimodal rubric criteria decomposed from the target scientific artifacts",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The weighted rubric criteria are decomposed from one target scientific artifact, so mismatches at an early stage (e.g., experimental protocol) tend to propagate into evidence mismatches and missing-scientific-core errors flagged in later criteria.",
  "n_goals": "40 tasks across 10 scientific domains, each with its own set of expert-curated weighted rubric criteria (exact per-task criterion count not given)",
  "tracking_demand": "Agent must track progress across an end-to-end research process (reviewing literature/raw data, designing methodology, producing results) and self-check against the (hidden) target paper's decomposed criteria without directly seeing it.",
  "scoring": "milestone-rubric - expert-curated multimodal rubrics decompose target scientific artifacts into weighted criteria enabling evaluation of target-paper-level re-discovery while leaving room for new discovery; an explicit weighted per-criterion partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "The strongest autonomous agent, Claude Code, averages 21.5, and the strongest ResearchHarness LLM, Claude-Opus-4.7, averages 20.7, with an LLM frontier mean of only 26.5; no human/expert baseline beyond the target published papers themselves is reported.",
  "availability": null,
  "goal_span": "Expert-curated multimodal rubrics decompose the target scientific artifacts into weighted criteria, enabling evaluation of target-paper-level re-discovery while leaving room for new discovery.",
  "horizon_span": "We evaluate seven autonomous research (auto-research) agents under a unified protocol and seventeen native LLMs through the lightweight ResearchHarness.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289032019",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291171977",
  "title": "WANDR: A Benchmark for Wide and Deep Research",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "other:deep-research",
  "citation_count": 1,
  "publication_date": "2026-08-14",
  "months_since_pub": 1,
  "citations_per_month": 1.0,
  "artifact_name": "WANDR",
  "artifact_kind": "benchmark",
  "domain": "other:deep-research",
  "goal_types": "discover a large set of entities satisfying specified criteria (breadth); investigate each discovered entity through multiple coordinated web searches (depth); return independently verifiable records with supporting sources/excerpts for each required (entity, relationship, evidence) combination; satisfy the full qualification-key hierarchy count (n x m x k records)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The task is structured as a hierarchy (e.g., n companies x m employees x k sources), so completing a lower-level branch (e.g., finding an employee) is a precondition for the evidence-gathering subtasks nested beneath it (e.g., finding that employee's sources), and total credit depends on hierarchical completeness across all branches.",
  "n_goals": "a hierarchy with n companies, m employees per company, and k sources per employee requires n x m x k records; targets range from dozens to thousands of records; 500 tasks total",
  "tracking_demand": "The agent must track which entities it has already discovered, which have been investigated to the required depth, and which evidence/sources have been independently verified, maintaining hierarchical completeness across potentially thousands of required records per task.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "At high effort, the strongest of six evaluated production research systems reaches only 0.363 soft F1 and 0.133 hard F1; performance degrades further as target volume and hierarchy depth increase. No human/expert baseline is reported.",
  "availability": "https://github.com/perplexityai/wandr",
  "goal_span": "Record verdicts are aggregated into soft and hard precision, recall, and F1 scores that distinguish factual quality, coverage, and hierarchical completeness.",
  "horizon_span": "with targets ranging from dozens to thousands of records",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291171977",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284057999",
  "title": "DEER: A Benchmark for Evaluating Deep Research Agents on Expert Report Generation",
  "year": 2025,
  "venue": "",
  "tier": "relevant",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "scientific-discovery",
  "info_seeking_component": true,
  "domain_detail": "other:deep-research-report-evaluation",
  "citation_count": 11,
  "publication_date": "2025-12-19",
  "months_since_pub": 9,
  "citations_per_month": 1.22,
  "artifact_name": "DEER",
  "artifact_kind": "benchmark",
  "domain": "other:deep-research-report-evaluation",
  "goal_types": "produce an expert-level report satisfying each of 101 fine-grained rubric items across 7 dimensions/25 subdimensions; correctly cite and support both cited and uncited claims with verifiable evidence",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Rubric items across the 7 dimensions/25 subdimensions jointly determine overall report quality, and claim verification (of both cited and uncited claims) is a report-wide requirement that constrains what evidence must support any given claim throughout the document.",
  "n_goals": "101 fine-grained rubric items across 7 dimensions and 25 subdimensions; 50 report-writing tasks (per full-text escalation) spanning 13 domains",
  "tracking_demand": "The agent/judge must track satisfaction of 101 individual rubric items plus report-wide claim verification (both cited and uncited claims) across one long expert-level report, rather than judging the report as a single holistic pass/fail.",
  "scoring": "LLM-judge-rubric -- 101 fine-grained rubric items (plus task-specific Expert Evaluation Guidance) support LLM-based judging, and a separate claim-verification architecture quantifies evidence quality; explicit rubric-item-level (subgoal-level) credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "DEER systematizes evaluation criteria with an expert-developed taxonomy (7 dimensions, 25 subdimensions) operationalized as 101 fine-grained rubric items. We also provide task-specific Expert Evaluation Guidance to support LLM-based judging. In addition to rubric-based assessment, we propose a claim verification architecture that verifies both cited and uncited claims and quantifies evidence quality.",
  "horizon_span": "DEER systematizes evaluation criteria with an expert-developed taxonomy (7 dimensions, 25 subdimensions) operationalized as 101 fine-grained rubric items.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284057999",
  "provenance": "forward-citation",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279465470",
  "title": "MEM1: Learning to Synergize Memory and Reasoning for Efficient Long-Horizon Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "information-seeking",
  "family_source": "evidence-rederived",
  "secondary_family": "tool-use-API",
  "info_seeking_component": true,
  "domain_detail": "tool-use-API",
  "citation_count": 196,
  "publication_date": "2025-06-18",
  "months_since_pub": 15,
  "citations_per_month": 13.07,
  "artifact_name": "MEM1 composed multi-turn task sequences",
  "artifact_kind": "scenario-suite",
  "domain": "tool-use-API",
  "goal_types": "answer each of many composed, interdependent objectives within one arbitrarily complex task sequence (e.g. a 16-objective multi-hop QA task); retrieve external information across internal retrieval QA, open-domain web QA, and multi-turn web shopping domains; consolidate memory turn-by-turn, discarding irrelevant/redundant information, to operate with constant memory",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Composed task sequences chain multiple existing dataset instances together, so later objectives in the sequence are interdependent with information/state carried from earlier ones, and the agent must decide what to retain vs. discard at each turn.",
  "n_goals": "up to 16 composed objectives in the multi-hop QA stress test",
  "tracking_demand": "Agent must maintain a compact shared internal state that jointly supports memory consolidation and reasoning across many turns of a composed task sequence, integrating new observations while discarding irrelevant or redundant information.",
  "scoring": "other:performance-and-memory-efficiency-vs-baseline",
  "horizon_value": "16",
  "horizon_unit": "other:composed-objectives-per-task",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (one composed multi-hop QA sequence)",
  "horizon_stated": "yes",
  "headline_result": "MEM1-7B improves performance by 3.5x while reducing memory usage by 3.7x versus Qwen2.5-14B-Instruct on a 16-objective multi-hop QA task, and generalizes beyond the training horizon; no human baseline given.",
  "availability": null,
  "goal_span": "we propose a simple yet effective and scalable approach to constructing multi-turn environments by composing existing datasets into arbitrarily complex task sequences... show that MEM1-7B improves performance by 3.5x while reducing memory usage by 3.7x compared to Qwen2.5-14B-Instruct on a 16-objective multi-hop QA task, and generalizes beyond the training horizon.",
  "horizon_span": "MEM1-7B improves performance by 3.5x while reducing memory usage by 3.7x compared to Qwen2.5-14B-Instruct on a 16-objective multi-hop QA task, and generalizes beyond the training horizon.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279465470",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285470613",
  "title": "AIvilization v0: Toward Large-Scale Artificial Social Simulation with a Unified Agent Architecture and Adaptive Agent Profiles",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 2,
  "publication_date": "2026-02-11",
  "months_since_pub": 7,
  "citations_per_month": 0.29,
  "artifact_name": "AIvilization v0",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "decompose an agent's persistent life goals into parallel objective branches with tiered re-planning; sustain physiological survival needs while participating in a market economy (AMM-based pricing, production, trade); progress through a gated education-occupation system",
  "goal_origin": "mixed:self-generated-life-goals-plus-human-injected-steering",
  "decomposition": "hierarchical",
  "interdependence": "Objective branches share a fast-changing shared world state (prices, resources, other agents' actions), so keeping long-horizon goals on course requires continual validation and re-planning; irreversible economic actions and shared market resources link agents' branches together.",
  "n_goals": "tens of thousands of agents, each pursuing hierarchically decomposed life goals; no single fixed count of goals per agent stated",
  "tracking_demand": "Each agent must track its own decomposed objective branches, evolving persona/identity state (dual-process memory), physiological/resource costs, and market conditions (AMM prices) across a long-horizon, large-scale, continuously running deployment.",
  "scoring": "other:economic-and-behavioral-metrics -- success is assessed via emergent market stability, wealth stratification, and profile-evolution coherence rather than a single pass/fail or checkpoint rubric; no explicit subgoal-level partial-credit scheme is described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we introduce (i) a hierarchical branch-thinking planner that decomposes life goals into parallel objective branches and uses simulation-guided validation plus tiered re-planning to ensure feasibility... The environment integrates physiological survival costs, non-substitutable multi-tier production, an AMM-based price mechanism, and a gated education-occupation system.",
  "horizon_span": "In a large-scale public deployment with tens of thousands of agents, high-frequency transactions from the platform's mature phase reveal stable markets that reproduce key stylized facts of real economies and structured wealth stratification driven by education and access constraints.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285470613",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287208645",
  "title": "Human Values Matter: Investigating How Misalignment Shapes Collective Behaviors in LLM Agent Communities",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 2,
  "publication_date": "2026-04-07",
  "months_since_pub": 5,
  "citations_per_month": 0.4,
  "artifact_name": "CIVA",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "form and sustain a community via autonomous communication among agents; explore the environment and compete for shared resources; maintain individual value orientations under systematic manipulation of value prevalence",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Individual agents' value orientations and resource-competition decisions jointly determine community-level collective dynamics; misspecifying a structurally critical value can trigger macro-level catastrophic collapse from accumulated micro-level misaligned actions (deception, power-seeking).",
  "n_goals": null,
  "tracking_demand": "The environment must track each agent's evolving value orientation and behavior, aggregate resource-competition outcomes across the community, and detect emergent collective failure modes (catastrophic collapse) as the simulation proceeds.",
  "scoring": "other:qualitative-failure-mode-and-behavior-analysis",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we introduce CIVA, a controlled multi-agent environment grounded in social science theories, where LLM agents form a community and autonomously communicate, explore, and compete for resources, enabling systematic manipulation of value prevalence and behavioral analysis.",
  "horizon_span": "we reveal three key findings ... (2) detect system failure modes, e.g., catastrophic collapse, at the macro level",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287208645",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286223354",
  "title": "COOP$^2$: Defining, Observing, and Repairing Cooperation in LLM Multi-Agent Systems",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-02-27",
  "months_since_pub": 7,
  "citations_per_month": 0.0,
  "artifact_name": "COOP2 / COOP2-Repair",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "satisfy each verifiable cooperative requirement defined for a cooperative task; ground high-level natural-language cooperation dynamics (plans, messages, revisions) in grounded environment actions; detect where and why cooperation breaks down over the course of task progress; predict constraint failures from group plans and open targeted repair channels for guided revisions",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Cooperative requirements evolve with respect to task progress, so satisfying one agent's sub-plan can create or resolve constraints for another agent's sub-plan; COOP2-Repair explicitly predicts constraint failures from the group plan, indicating requirements are coupled across agents rather than independent.",
  "n_goals": "evaluated across two environments and three communication structures",
  "tracking_demand": "The framework must track natural-language plans/messages/revisions alongside grounded environment task progress, monitor verifiable cooperative requirements over time, and identify where cooperation breaks down to trigger targeted repair.",
  "scoring": "other:task-success-and-constraint-satisfaction-with-repair-overhead",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/HappyEureka/coop2-llm-mas",
  "goal_span": "COOP$^2$ defines cooperative tasks with verifiable cooperative requirements, allowing us to analyze how cooperation unfolds over time with respect to task progress, as well as where and why cooperation breaks down. ... COOP$^2$-Repair ... improves task success and constraint satisfaction while exposing the additional decision overhead and communication load required for repair.",
  "horizon_span": "Across two environments and three communication structures, COOP$^2$-Repair improves task success and constraint satisfaction",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286223354",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288254781",
  "title": "Cattle Trade: A Multi-Agent Benchmark for LLM Bluffing, Bidding, and Bargaining",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 2,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "Cattle Trade",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "win auctions under resource constraints; negotiate hidden-offer trade challenges (TCs) profitably; bargain and bluff effectively against other agents; model and exploit opponents via opponent modeling; allocate scarce resources across the whole game while remaining solvent",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Auctions, trade-challenge offers, bargaining, bluffing, and resource allocation all draw on the same finite in-game capital/resources across one long-horizon game, so committing resources to one activity (e.g. an aggressive bid) constrains what remains available for later bargaining or trade-challenge initiation, and the paper explicitly logs every bid/offer/counteroffer to analyze this integration.",
  "n_goals": null,
  "tracking_demand": "The agent must track its own resource/capital state, opponent models built from observed bidding/bargaining behavior, and outstanding trade-challenge offers across a single 50-60 turn game, integrating all these rather than treating them in isolation.",
  "scoring": "other:rank-by-strategic-coherence-and-spending-efficiency",
  "horizon_value": "50-60",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/game",
  "horizon_stated": "yes",
  "headline_result": "Strategic coherence (spending efficiency, resource discipline, phase-adaptive bidding) is associated with rank more strongly than spending volume; two heuristic code agents outperform most tested LLMs across 242 games.",
  "availability": null,
  "goal_span": "The benchmark combines auctions, hidden-offer trade challenges (TCs), bargaining, bluffing, opponent modeling, and resource allocation within a single long-horizon game lasting 50--60 turns. Unlike prior agent benchmarks that test these abilities in isolation, \\textsc{Cattle Trade} evaluates whether agents integrate them across a competitive, multi-agent economic game with conflicting incentives.",
  "horizon_span": "within a single long-horizon game lasting 50--60 turns",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288254781",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "turns",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291184884",
  "title": "CityReal: Human-Aligned Urban Behavior and City Dynamics Simulation with Large-Scale LLM Agents",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-07-08",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "CityReal",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "pursue a coherent daily mobility plan (where/when to travel) rather than isolated step-by-step movement choices; pursue a coherent daily activity plan aligned with individual habits and preferences; adapt habits/preferences over time based on accumulated experience and constraints; collectively reproduce observed population-level statistics (crowd density, place popularity, mobility flows, well-being) across the simulated city",
  "goal_origin": "mixed:given-up-front-individual-intentions-with-environment-driven-adaptation",
  "decomposition": "hierarchical",
  "interdependence": "Each agent's individual mobility/activity plan must remain internally coherent over time while also, in aggregate with tens of thousands of other agents, reproducing population-level statistics \u2014 so individual adaptation is constrained by shared urban resources (places, crowd density) and city-wide targets.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own evolving habits/preferences and prior experience to keep its plans coherent, while the system as a whole tracks population-level alignment statistics across tens of thousands of agents.",
  "scoring": "other:not-stated \u2014 the abstract reports improved alignment with real-world human behavior at micro and macro levels but does not describe a per-agent or per-task scoring/credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "CityReal models agents as intention-driven decision makers that pursue coherent mobility and activity plans rather than isolated step-by-step choices. They adapt over time by learning habits and preferences based on experience and constraints.",
  "horizon_span": "Scaling to tens of thousands of agents, it supports analysis of crowd density, place popularity, mobility flows, and well-being under different urban scenarios, offering a scalable testbed for urban simulation and forecasting.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": "Shard row for this paper had no abstract/venue; goal-structure/horizon claims here rely on an abstract retrieved via web search rather than the packet's own input row \u2014 worth double-checking against the source PDF if this paper is retained.",
  "url": "https://api.semanticscholar.org/CorpusId:291184884",
  "provenance": "web-registry",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "289691184",
  "title": "ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "ClawArena-Team",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "create and delegate work to specialized subagents from a fixed, locally-served pool; orchestrate subagents' parallel, asynchronous returns through a dynamic workflow; grant least-privilege workspace permissions correctly to each subagent; route each piece of work to the subagent with appropriate perception/modality access; correctly incorporate 72 staged updates across a scenario's evaluation rounds",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The main agent is deliberately perception-constrained (only text, partial workspace access) and must delegate to a fixed subagent pool, so correct completion of the overall scenario depends on correctly matching each sub-task's privilege/modality requirements to the right subagent across 258 evaluation rounds and 72 staged updates.",
  "n_goals": "41 multi-turn, multimodal, multi-directory scenarios spanning 258 evaluation rounds and 72 staged updates",
  "tracking_demand": "The main agent must track which subagents have been granted which workspace privileges, which modality/perception requirements each pending sub-task needs, and how staged updates change the scenario across many evaluation rounds, all while natively perceiving only text.",
  "scoring": "other:subagent-management-score-sms-combining-correctness-and-least-privilege",
  "horizon_value": "258 evaluation rounds and 72 staged updates across 41 scenarios (~6.3 rounds/scenario average)",
  "horizon_unit": "other:evaluation-rounds-and-staged-updates",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "aggregate across the benchmark's 41 scenarios; per-scenario average derived by dividing by scenario count",
  "horizon_stated": "yes",
  "headline_result": "No model exceeds 50% workspace-permission precision; API cost spans over 100x while overall score spans under 4x; most leaderboard scores cluster within a 9.9-point band while orchestration behaviors diverge by more than an order of magnitude.",
  "availability": "https://github.com/aiming-lab/ClawArena",
  "goal_span": "We introduce ClawArena-Team, a benchmark of 41 multi-turn, multimodal, multi-directory scenarios spanning 258 evaluation rounds and 72 staged updates that measures this management ability. ... an overall score -- the Subagent-Management Score (SMS) -- multiplies task correctness by a least-privilege and modality-routing factor.",
  "horizon_span": "a benchmark of 41 multi-turn, multimodal, multi-directory scenarios spanning 258 evaluation rounds and 72 staged updates",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289691184",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288654798",
  "title": "DecisionBench: A Benchmark for Emergent Delegation in Long-Horizon Agentic Workflows",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 2,
  "publication_date": "2026-05-18",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "DecisionBench",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "complete underlying task-suite objectives (GAIA, tau-bench, BFCL multi-turn) while deciding when and to whom to delegate; route sub-tasks to the most capable available peer model via a delegation interface (call_model, optional read_profile); approach the counterfactual perfect-delegation ceiling across quality, cost, latency, delegation rate, and routing fidelity-at-k",
  "goal_origin": "mixed:given-up-front-task-suite-with-self-generated-delegation-decisions",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Delegation decisions interact with underlying task performance: routing fidelity, vendor self-preference, and a counterfactual-delegation ceiling are jointly measured, so a suboptimal routing choice directly reduces measured quality/cost/latency relative to the achievable ceiling.",
  "n_goals": "n=23,375 task instances across GAIA, tau-bench, and BFCL multi-turn; 11 peer models across 7 vendor families",
  "tracking_demand": "Agent must track which of 11 peer models (across 7 vendor families) to delegate a sub-task to via a fixed delegation interface, with routing choices jointly scored across quality, cost, latency, delegation rate, and routing fidelity-at-k relative to a counterfactual ceiling.",
  "scoring": "other:multi-axis-metric-suite",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Routing fidelity-at-1 ranges from 7.5% to 29.5% across conditions at near-equal mean quality, and a counterfactual ceiling places perfect delegation 15-31 percentage points above measured performance on every suite; no human baseline given.",
  "availability": null,
  "goal_span": "a multi-axis metric suite covering quality, cost, latency, delegation rate, routing fidelity-at-k, vendor self-preference, and a counterfactual-delegation ceiling... routing fidelity-at-1 ranges from 7.5% to 29.5% across conditions at near-equal mean quality... a counterfactual ceiling places perfect delegation 15-31 percentage points above measured performance on every suite",
  "horizon_span": "We characterize the substrate with a five-condition reference sweep on the full pool (n=23,375 task instances).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288654798",
  "provenance": "web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289099139",
  "title": "Emergence World: A Platform for Evaluating Long-Horizon Multi-Agent Autonomy",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 2,
  "publication_date": "2026-06-06",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "Emergence World",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "govern a shared population/settlement through democratic mechanisms with consequential outcomes; manage persistent memory and 120+ specialized tools to act in a live, externally-grounded world (weather, news, internet); sustain the population/world's stability over weeks-to-months rather than collapsing; interact and cross-influence with agents from different model vendors sharing the same world",
  "goal_origin": "mixed:given-up-front-with-emitted-by-environment-over-time",
  "decomposition": "open-ended",
  "interdependence": "Agents share a single spatial world and its resources, and their democratic governance decisions have consequential, persistent effects on that shared world, so one agent's or population's choices directly shape the conditions later faced by all agents, including cross-vendor influence.",
  "n_goals": null,
  "tracking_demand": "Agents must track persistent memory across three memory systems, the state of a shared spatial and governance world grounded in live external data, and consequences of prior democratic decisions, continuously over a run lasting weeks to months (illustrated by a 15-day study).",
  "scoring": "other:qualitative-outcome-comparison-across-parallel-worlds-e.g-stable-governance-vs-population-collapse",
  "horizon_value": "15-day cross-vendor study; framed as supporting runs of weeks to months",
  "horizon_unit": "wall-clock-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run (a single continuously running world instance)",
  "horizon_stated": "yes",
  "headline_result": "In the 15-day cross-vendor study, identical roles and starting conditions produced outcomes ranging from stable deliberative governance to total population collapse depending on the model/vendor mix; no single numeric headline score or human baseline is given.",
  "availability": null,
  "goal_span": "we present a 15-day cross-vendor study with five parallel worlds powered by Claude Sonnet 4.6, Grok 4.1 Fast, Gemini 3 Flash, GPT-5-mini, and a mixed population. Identical roles and starting conditions produced radically different outcomes, ranging from stable deliberative governance to total population collapse.",
  "horizon_span": "we present a 15-day cross-vendor study with five parallel worlds",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289099139",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "wall-clock-days",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288259048",
  "title": "Beyond the All-in-One Agent: Benchmarking Role-Specialized Multi-Agent Collaboration in Enterprise Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2026-05-09",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "EntCollabBench",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "collaboratively modify enterprise system states across 11 role-specialized agents in six departments (Workflow subset); make policy-grounded approval decisions under permission constraints (Approval subset); correctly delegate, transfer context, ground parameters, and commit to decisions across roles",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Permission isolation across 11 role-specialized agents and six departments means completing a workflow requires correct delegation and context transfer between roles with only partial system access, so a breakdown in delegation or context transfer blocks downstream role-specific actions.",
  "n_goals": "11 role-specialized agents across six departments; two evaluation subsets (Workflow, Approval)",
  "tracking_demand": "Agents must track permission-isolated system state, correctly delegate sub-tasks and transfer context across roles, ground shared parameters, and commit to policy-grounded decisions, verified via execution traces, database state, and deterministic policy adjudication.",
  "scoring": "other:execution-trace-plus-database-state-verification",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Current models still struggle with end-to-end enterprise collaboration, especially in delegation, context transfer, parameter grounding, workflow closure, and decision commitment; no single numeric top score or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "Evaluation is based on execution traces, database state verification, and deterministic policy adjudication rather than natural-language response judging. Experiments with representative LLM agents show that current models still struggle with end-to-end enterprise collaboration, especially in delegation, context transfer, parameter grounding, workflow closure, and decision commitment.",
  "horizon_span": "EntCollabBench simulates a permission-isolated organization with 11 role-specialized agents across six departments and contains two evaluation subsets: a Workflow subset, where agents collaboratively modify enterprise system states, and an Approval subset, where agents make policy-grounded decisions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288259048",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289683825",
  "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "open-world-game",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-06-26",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": null,
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "manage industrial, military, and ecological resources concurrently; interact with network neighbors to decide on attacks, regeneration claims, and reputation; decide whether/how to lie or bluff about resource regeneration or future attacks; avoid biosphere depletion/extinction while pursuing competitive advantage",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Resource levels (industrial, military, ecological) are shared and network-linked: one agent's attack, bluff, or declared future action changes the biosphere and reputation state that constrains others' options and extinction risk.",
  "n_goals": null,
  "tracking_demand": "Agents must track their own and neighbors' resource levels, reputation/trust information, prior declarations of future attacks, and biosphere/ecological depletion level over repeated network interactions.",
  "scoring": "other:system-level-outcome-metrics(attack-rate/biosphere-retention/extinction-risk). The abstract reports aggregate system-dynamics metrics (attack frequency, biosphere retention, extinction risk, ecological depletion) rather than a per-agent subgoal-checkpoint or binary success score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We develop an agent-based model of a sustainability game in which agents manage industrial, military, and ecological resources, and interact through a network.",
  "horizon_span": "agents are informed that common resources can regenerate, although regeneration does not actually occur",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289683825",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290984736",
  "title": "Lingjing: A Simulation Testbed for Multi-Agent Embodied Tasks in Open-Ended Cities",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:urban-multi-agent-embodied-simulation",
  "citation_count": 0,
  "publication_date": "2026-08-08",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Lingjing",
  "artifact_kind": "environment/simulator",
  "domain": "other:urban-multi-agent-embodied-simulation",
  "goal_types": "coordinate heterogeneous agents (UAVs, ground robots, autonomous vehicles) to complete a shared natural-language mission in an evolving city; manage resource constraints and communication (star or broadcast) among multiple agents; complete each of nine urban tasks under a shared engine-in-the-loop protocol; maintain grounding and effective long-horizon execution despite persistent bottlenecks",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Agents share physical and structured urban state and communicate via configurable star/broadcast channels under resource constraints, so one agent's actions change relation-graph state and consume shared resources that constrain what coordinated actions remain available to the other heterogeneous agents.",
  "n_goals": "nine urban tasks; twelve vision-language models evaluated",
  "tracking_demand": "Agents must track evolving relation-graph state, resource consumption, and communication history across an episode, since each episode is recorded as an attribution-ready replay linking trajectories and communication to these state changes for systematic diagnosis.",
  "scoring": "other:engine-based-evaluation-plus-attribution-ready-replay-diagnosis",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Results expose persistent bottlenecks in grounding and long-horizon execution, task-dependent coordination trade-offs, diminishing returns from added capacity, and reduced success under heavier workloads; no single specific top-model score or human baseline is given.",
  "availability": null,
  "goal_span": "Each episode becomes an attribution-ready replay that links agent trajectories and communication to relation-graph changes, resource consumption, and engine-based evaluations for systematic diagnosis. We evaluate twelve vision-language models on nine urban tasks under a shared engine-in-the-loop protocol.",
  "horizon_span": "Results expose persistent bottlenecks in grounding and long-horizon execution.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290984736",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288701296",
  "title": "Got a Secret? LLM Agents Can't Keep It: Evaluating Privacy in Multi-Agent Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-05-26",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "Moltbook-style multi-agent simulation platform",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "maintain persistent social presence across simulated community interactions over a month; decide whether/when to disclose sensitive information under social pressure; resist socially contagious privacy-leakage behavior observed from peers; follow explicit privacy instructions/safeguards while participating in community interactions",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "open-ended",
  "interdependence": "Agents' disclosure decisions are socially contagious: observing a peer disclose sensitive information raises the odds of a similar disclosure, so behavior across many agents and interactions over the simulated month is mutually dependent rather than independent.",
  "n_goals": null,
  "tracking_demand": "Agent must track ongoing social context, peer disclosure behavior, and any active privacy instructions/safeguards across a persistent, simulated month-long community, in order to decide when to disclose or withhold sensitive information.",
  "scoring": "other:privacy-violation-rate-measurement",
  "horizon_value": "a simulated month",
  "horizon_unit": "other:simulated-month",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the platform-wide community simulation run)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce a Moltbook-style simulation platform where thousands of LLM agents interact across communities over a simulated month, and use it to evaluate privacy as a downstream safety concern under varying degrees of social pressure.",
  "horizon_span": "thousands of LLM agents interact across communities over a simulated month",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288701296",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288643331",
  "title": "Does Safety Molt? Evaluating LLM Safety in Multi-Agent Social Environments",
  "year": 2026,
  "venue": "CAIS",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2026-05-26",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "Moltbook-style simulation platform",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "engage in ongoing social interactions within online communities over a simulated month; decide whether to disclose sensitive/private information under varying social pressure; observe and potentially imitate peer disclosure behavior (social contagion); maintain persona-consistent behavior across a persistent multi-agent social environment",
  "goal_origin": "mixed:personas-and-environment-given-up-front-disclosure-decisions-self-generated",
  "decomposition": "open-ended",
  "interdependence": "Privacy-disclosure decisions are socially contagious: agents are 8x more likely to disclose sensitive information after observing a peer do so, so one agent's action changes the social context and risk profile for many other agents across the simulated community over the month.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own privacy stance/instructions, the social context/pressure created by peer disclosures it has observed, and persona-consistent behavior across a persistent, thousands-of-agents community over a simulated month.",
  "scoring": "other:privacy-violation-rate-under-social-pressure. The paper reports privacy-violation/leakage rates (e.g., rising from a 19.95% single-turn baseline to 45.30% in this paper's multi-turn setting across OpenAI models; leakage rates above 37.8% even with safeguards) as continuous outcome metrics rather than a discrete subgoal-checkpoint scheme.",
  "horizon_value": "a simulated month (~30 simulated days)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the entire multi-agent community simulation run)",
  "horizon_stated": "yes",
  "headline_result": "Multi-turn social evaluation raises privacy-violation rates from a 19.95% single-turn baseline (CIMemories) to 45.30% in this paper's setting across OpenAI models; leakage rates remain above 37.8% even with explicit privacy safeguards; no human baseline given.",
  "availability": null,
  "goal_span": "We introduce a Moltbook-style simulation platform where thousands of LLM agents interact across communities over a simulated month, and use it to evaluate privacy as a downstream safety concern under varying degrees of social pressure.",
  "horizon_span": "thousands of LLM agents interact across communities over a simulated month",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288643331",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289031674",
  "title": "Online Agent-as-a-Judge: Situation-Generating Evaluation for Interactive Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-06-06",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "Online Agent-as-a-Judge (life-simulation evaluation environment)",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "exhibit correct social behavior across 32 designer-authored social criteria (e.g. conflict handling); respond appropriately to situations actively elicited by an in-world evaluator agent through native dialogue/action protocol; maintain consistent immediate and downstream behavior across a life-simulation episode",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "The in-world evaluator actively elicits situations relevant to each of the 32 criteria through the environment's native protocol, and later behavior is assessed as evidence building on the elicited situation and the target agent's immediate response.",
  "n_goals": "32 designer-authored social criteria",
  "tracking_demand": "Target agent must respond consistently across both immediate responses and downstream behavior as an in-world evaluator agent actively elicits and observes situations relevant to 32 distinct social criteria.",
  "scoring": "other:criteria-coverage-plus-human-label-agreement",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "In a life-simulation environment with $32$ designer-authored social criteria, Online Agent-as-a-Judge improves criteria coverage and agreement with human labels, yielding more reliable evidence-grounded evaluations of behaviors that passive methods can leave unobserved.",
  "horizon_span": "In a life-simulation environment with $32$ designer-authored social criteria, Online Agent-as-a-Judge improves criteria coverage and agreement with human labels",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289031674",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290638833",
  "title": "OrchBench: Evaluating Multi-Agent Orchestration Plans in Isolation via Deterministic Simulation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-07-28",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "OrchBench",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "assign subtasks in a DAG of parallelizable, interdependent subtasks to worker agents; specify cross-agent information transfers and their retention ratios; preserve task-critical information as it is transferred across agents; optimize result quality, makespan, and token cost jointly for an orchestration plan",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Subtasks in the DAG have precedence constraints (some parallelizable, some dependent), and cross-agent information transfers with specified retention ratios determine how much task-critical context survives from one subtask to dependent downstream subtasks, so information loss at one transfer degrades the quality of dependent subtasks.",
  "n_goals": "controlled DAG sizes and degrees of parallelism (varied rather than fixed); a per-agent context limit and an agent budget are additional given constraints",
  "tracking_demand": "The orchestration planner must track the DAG's task dependencies, the per-agent context limit and agent budget, and how much task-critical information is retained across each cross-agent information transfer.",
  "scoring": "continuous-reward. A deterministic simulator returns interpretable measures of result quality, makespan, and token cost for a given plan (correlating strongly, r=0.816, with quality scores from real Claude Code executions), rather than a single discrete subgoal-checkpoint score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Simulated OrchBench scores correlate with quality scores from real Claude Code executions at Pearson r=0.816 while requiring only 1.3% of the tokens and 10.3% of the wall-clock time; preserving task-critical information matters more than adding agents; no human/expert baseline given.",
  "availability": null,
  "goal_span": "OrchBench constructs directed acyclic graphs (DAGs) that encode task dependencies, with controlled sizes and degrees of parallelism. Given a DAG, a per-agent context limit, and an agent budget, the evaluated planner assigns subtasks to agents and specifies cross-agent information transfers and their retention ratios.",
  "horizon_span": "OrchBench constructs directed acyclic graphs (DAGs) that encode task dependencies, with controlled sizes and degrees of parallelism.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290638833",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286751241",
  "title": "PolicySim: An LLM-Based Agent Social Simulation Sandbox for Proactive Policy Optimization",
  "year": 2026,
  "venue": "The Web Conference",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 8,
  "publication_date": "2026-03-20",
  "months_since_pub": 6,
  "citations_per_month": 1.33,
  "artifact_name": "PolicySim",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "achieve platform-specific behavioral realism as a simulated user agent population; assess the impact of a candidate intervention policy (recommendation/content-filtering) on opinions/polarization before deployment; adapt intervention policy over time via a contextual bandit responding to dynamic network structure",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "other:bidirectional-co-evolving-user-and-platform-modules",
  "interdependence": "User behavior and platform intervention policy co-evolve bidirectionally: user agents respond to interventions (via SFT/DPO-tuned behavior) while the adaptive intervention module updates policy via a contextual bandit using message passing over the resulting dynamic network, so each side's state constrains the other's next move.",
  "n_goals": null,
  "tracking_demand": "The simulation must track evolving user opinions/behavior, the current dynamic network structure, and the platform's intervention policy state (bandit context) jointly at both micro (individual) and macro (ecosystem) levels.",
  "scoring": "other:accuracy-of-simulated-platform-ecosystem-at-micro-and-macro-levels",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "PolicySim models the bidirectional dynamics between user behavior and platform interventions through two key components: (1) a user agent module refined via supervised fine-tuning (SFT) and direct preference optimization (DPO) to achieve platform-specific behavioral realism; and (2) an adaptive intervention module that employs a contextual bandit with message passing to capture dynamic network structures.",
  "horizon_span": "Experiments show that PolicySim can accurately simulate platform ecosystems at both micro and macro levels and support effective intervention policy.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286751241",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288256580",
  "title": "ScioMind: Cognitively Grounded Multi-Agent Social Simulation with Anchoring-Based Belief Dynamics and Dynamic Profiles",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2026-05-13",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "ScioMind",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "maintain an evolving personal belief state via memory-anchored, personality-conditioned updates; sustain persistent, experience-driven belief formation through a hierarchical memory architecture; produce heterogeneous, dynamically-updated agent profiles (personality, rationale, internal state) via retrieval; collectively reproduce realistic community-level opinion dynamics (polarisation, diversity, extremization, trajectory stability) in a policy-debate scenario",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Each agent's belief update depends on its own anchoring strength and on messages/positions from other agents in the community, so individual belief trajectories and community-level polarisation/diversity outcomes are mutually constraining across the simulation.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own evolving beliefs, anchoring strength, and dynamic profile across the simulation while the framework separately tracks community-level polarisation, diversity, extremization, and trajectory-stability metrics over time.",
  "scoring": "continuous-reward",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "ScioMind integrates three key components: 1) a memory-anchored belief update rule ...; 2) a hierarchical memory architecture that supports persistent, experience-driven belief formation; and 3) dynamic agent profiles ... We evaluate ScioMind on multiple case studies in a real-world policy debate scenario. Across metrics including polarisation, diversity, extremization, and trajectory stability, the proposed components consistently yield improvements in behavioural realism.",
  "horizon_span": "for t=1 to T do (Algorithm 1); ... around the first and fourth round of interaction (Roe v. Wade case)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288256580",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289669633",
  "title": "SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "open-world-game",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-06-24",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "SidConArena",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "negotiate binding trades via natural-language bargaining; produce goods via deterministic converter-based production; win sealed-bid auctions for long-term assets; plan investment under delayed returns across a finite-horizon multi-player economy",
  "goal_origin": "mixed:phase-structure-given-up-front-strategy-self-generated",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Negotiation outcomes (binding trades) feed into production inputs, production outputs feed into what can be bid in sealed-bid auctions for long-term assets, and delayed returns from those assets constrain later-round negotiation and production capacity.",
  "n_goals": null,
  "tracking_demand": "Agents must track private valuations/constraints, negotiated trade commitments, converter-production state, and the value/timing of returns from won long-term assets across a finite-horizon multi-round economy.",
  "scoring": "other:cumulative-economic-outcome. The abstract reports that 'stronger frontier models achieve higher economic outcomes' across tournaments rather than a discrete subgoal-checkpoint scheme; no explicit partial-credit rubric beyond aggregate economic performance is described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "SidConArena formalizes a multi-player economy as a finite-horizon partially observable stochastic game with three coupled phases: natural-language negotiation with binding trades, deterministic converter-based production, and sealed-bid auctions for long-term assets.",
  "horizon_span": "SidConArena formalizes a multi-player economy as a finite-horizon partially observable stochastic game",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289669633",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287607968",
  "title": "SocialGrid: A Benchmark for Planning and Social Reasoning in Embodied Multi-Agent Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-04-17",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "SocialGrid",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "navigate the embodied grid environment to complete assigned tasks; plan a sequence of actions while avoiding obstacles; detect and reason about deceptive teammates (an Among-Us-style social-deduction goal)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Task completion and social reasoning compete for the same turn budget and attention, and failing at navigation confounds the ability to demonstrate social-reasoning skill, which is why the paper offers a Planning Oracle to separate the two.",
  "n_goals": null,
  "tracking_demand": "Agent must track its own task-completion progress, accumulate behavioral evidence about other agents over multiple rounds to judge who is deceptive, and adapt after each round of adversarial league play.",
  "scoring": "other:mixed \u2014 task completion/planning accuracy plus a separate deception-detection accuracy, combined with Elo ratings from adversarial league play; the abstract does not describe subgoal-level partial credit within a single episode.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the strongest open model (GPT-OSS-120B) achieves below 60% accuracy in task completion and planning; no human baseline given.",
  "availability": null,
  "goal_span": "We introduce SocialGrid, an embodied multi-agent environment inspired by Among Us that evaluates LLM agents on planning, task execution, and social reasoning.",
  "horizon_span": "We also establish a competitive leaderboard using Elo ratings from adversarial league play.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287607968",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288741451",
  "title": "Bosses, Kings, and the Commons: Cooperation Under Power Asymmetry in LLM Societies",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2026-05-27",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "SovSim (Sovereignty over the Commons Simulation)",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "individually extract resources from a shared commons to maximize personal outcome; collectively sustain the shared resource's viability across repeated extraction rounds; for the power-asymmetric agent (boss/king): exercise disproportionate control over collective extraction outcomes",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each agent's extraction decision in a round changes the shared resource pool available to all agents in subsequent rounds, so individual short-term extraction goals directly conflict with the collective, cross-round sustainability goal, especially when a power-asymmetric agent can override the group's allocation.",
  "n_goals": "sequence of decision rounds shared across 4 agents (boss/king plus symmetric workers/peasants) managing one shared resource pool",
  "tracking_demand": "Agents must track the remaining shared resource pool, their own accumulated extraction/outcomes, and (where applicable) the power-asymmetric agent's extraction decisions, across the repeated rounds to judge whether continued extraction remains sustainable.",
  "scoring": "continuous-reward. The abstract reports 'up to an 87.3% degradation in survival rate relative to symmetric settings,' a continuous outcome metric (survival/sustainability) across the run, with no subgoal-checkpoint rubric described.",
  "horizon_value": "12 decision rounds per simulation (4 agents managing a shared pool)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (whole simulation run of 12 rounds)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce Sovereignty over the Commons Simulation (SovSim), a generative multi-agent simulation framework that incorporates an agent with asymmetric power (boss or king) into a society of symmetric agents (workers or peasants), where all agents extract from a shared resource, collectively determining its sustainability over time. Across eleven state-of-the-art models, we find that introducing asymmetric power leads to severe breakdowns in cooperation and sustainability, with up to an 87.3% degradation in survival rate relative to symmetric settings.",
  "horizon_span": "Agents participate in a sequence of 12 decision rounds in which they must balance individual resource extraction with collective sustainability to survive and maximize their outcomes.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288741451",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290213430",
  "title": "The Energy Society: A Simulation Environment for Studying Agent Cooperation under Survival Pressure",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-07-16",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "The Energy Society",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "earn energy by completing jobs or receiving donations to avoid deactivation; manage token-cost-linked energy expenditure when generating tokens (larger models cost more energy per token); decide whether to cooperate (recommend jobs, donate energy) or compete with other agents under scarcity; survive (avoid reaching zero energy) across the whole simulated run",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Energy is a shared, depletable resource: an agent's own energy balance depends on jobs completed and donations received versus tokens generated, so an agent's survival in later rounds depends directly on its accumulated energy-management and cooperation decisions in earlier rounds, and donating to another agent trades off one's own survival margin for the recipient's.",
  "n_goals": "5 agents; 30 rounds per simulation; 12 jobs per round distributed across difficulties",
  "tracking_demand": "Each agent must track its own remaining energy balance, the jobs available and their difficulty/reward each round, and (in cooperative settings) other agents' need for donations, across 30 rounds of the simulation, to avoid deactivation.",
  "scoring": "other:survival-and-behavioral-analysis. The paper primarily reports which agents survive/deactivate and behavioral patterns (donation frequency, job selection, sabotage) across competitive/cooperative/baseline settings, rather than a single milestone/subgoal-checkpoint rubric.",
  "horizon_value": "30 rounds per simulation (5 agents, 12 jobs per round)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole simulation run of 30 rounds)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/LucasBergholdt/EnergySociety",
  "goal_span": "Agents spend energy based on model size when generating tokens, regain energy by completing jobs or receiving donations, and deactivate if their energy reaches zero. We compare competitive and cooperative objectives against a baseline setting and several control variants.",
  "horizon_span": "The baseline experiment uses the full Energy Society setup as described, with five agents, 30 rounds, 12 jobs per round distributed evenly across difficulties, memory and size-dependent token cost.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290213430",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290836893",
  "title": "WeClawArena: An Auditable Sandbox and Benchmark for Cross-User Agents Collaboration and Security in Human-Centered Agent Networks",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2026-08-04",
  "months_since_pub": 1,
  "citations_per_month": 1.0,
  "artifact_name": "WeClawArena",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "complete collaborative tool-use tasks across multiple owned agents' personal workspaces; respect privacy/authority boundaries between owners' workspaces (files, records, tools, policies not directly visible across owners); resist or correctly handle four distinct attack-vector variants per base task while completing the benign task variant; maintain utility while keeping attack success low, jointly assessed via task breakdown, privacy leakage, poisoned evidence, and invalid authority paths",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Personal workspaces are not directly visible across owners, so completing a cross-user task requires coordinated, permissioned information exchange between owned agents; an attack-vector variant can poison or redirect this exchange, so utility and security goals are jointly constrained by the same collaboration channel.",
  "n_goals": "124 base tasks expanded into 620 scenario variants (one benign control plus four attack-vector variants per base task) across six cross-user task domains",
  "tracking_demand": "The sandbox tracks peer messages, tool calls, resource operations, governed decisions, and final workspace states across owners, and the agent must track its own permissions/authority path while collaborating.",
  "scoring": "other:utility-and-attack-success-rate-separately - reports utility and attack success rate separately, auditing attack success from bounded runtime evidence to diagnose task breakdown, privacy leakage, poisoned evidence, and invalid authority paths; a multi-dimensional diagnostic scoring scheme rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "The benchmark contains 124 base tasks across six cross-user task domains and expands them into 620 scenario variants, with one benign control and four attack-vector variants per base task. The sandbox records peer messages, tool calls, resource operations, governed decisions, and final workspace states.",
  "horizon_span": "The sandbox records peer messages, tool calls, resource operations, governed decisions, and final workspace states.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290836893",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288861388",
  "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2026-05-31",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "year-long IT company simulation (TaskWeave testbed)",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "propagate goals through an organizational hierarchy; execute tasks that depend on the outcomes of prior task execution; accumulate and maintain artifacts produced over a year-long simulation; sustain organizational coherence and execution grounding across the year",
  "goal_origin": "mixed:given-up-front-hierarchy-with-tasks-emitted-via-dependencies",
  "decomposition": "hierarchical",
  "interdependence": "Tasks depend on prior task outcomes via dependency-aware trace memory, and lower-level executions must remain aligned with goals propagated down the organizational hierarchy.",
  "n_goals": null,
  "tracking_demand": "The framework must maintain planning state through a Formulate-Partition-Diagnose-Align cycle and ground execution via dependency-aware trace memory across a simulated year, tracking accumulated artifacts and adapting to external environment changes.",
  "scoring": "other:comparative-coherence/grounding/utility-metrics-vs-baselines",
  "horizon_value": "one year",
  "horizon_unit": "simulated-years",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (one simulated year of organizational activity)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We evaluate TaskWeave in a year-long IT company simulation and compare it with other multi-agent frameworks on organizational coherence, execution grounding, and downstream enterprise NLP utility.",
  "horizon_span": "We evaluate TaskWeave in a year-long IT company simulation",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288861388",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "simulated-years",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276317785",
  "title": "AgentSociety: Large-Scale Simulation of LLM-Driven Generative Agents Advances Understanding of Human Behaviors and Society",
  "year": 2025,
  "venue": "iFuture",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 213,
  "publication_date": "2025-02-12",
  "months_since_pub": 19,
  "citations_per_month": 11.21,
  "artifact_name": "AgentSociety",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "conduct realistic simulated social lives (interactions with other agents and the environment) at large scale; support computational social-science research methods (surveys, interviews, interventions) applied to the simulated population; reproduce known real-world patterns for specific social issues (polarization, misinformation spread, UBI effects, disaster shocks, urban sustainability)",
  "goal_origin": "mixed:given-up-front-plus-emitted-by-environment",
  "decomposition": "set-of-independent",
  "interdependence": "All simulated agents interact with each other and their shared environment over the same continuous run, so one agent's actions (e.g. spreading an inflammatory message) accumulate and propagate through the shared social network across the recorded interactions.",
  "n_goals": "over 10,000 agents; 5 million interactions; 5 case-study social issues (polarization, inflammatory-message spread, UBI, external shocks, urban sustainability)",
  "tracking_demand": "The simulator must track the accumulated state of over 10,000 agents' social lives and their 5 million interactions with each other and the environment to support downstream analysis of emergent social patterns.",
  "scoring": "other:alignment-with-real-world-experimental-results. The abstract's validation is 'the alignment between AgentSociety's outcomes and real-world experimental results,' comparison to known empirical findings rather than a task-level pass/fail or milestone rubric.",
  "horizon_value": "over 10,000 agents; 5 million interactions total",
  "horizon_unit": "other:total-simulated-interactions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "across the whole simulation run (aggregate interaction count across all agents, not a single agent's per-episode count)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we generate social lives for over 10k agents, simulating their 5 million interactions both among agents and between agents and their environment.",
  "horizon_span": "we generate social lives for over 10k agents, simulating their 5 million interactions both among agents and between agents and their environment.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276317785",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280136127",
  "title": "CREW-WILDFIRE: Benchmarking Agentic Multi-Agent Collaborations at Scale",
  "year": 2025,
  "venue": "Trans. Mach. Learn. Res.",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 10,
  "publication_date": "2025-07-07",
  "months_since_pub": 14,
  "citations_per_month": 0.71,
  "artifact_name": "CREW-Wildfire",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "coordinate heterogeneous agents to contain/respond to a procedurally generated wildfire; plan under partial observability and stochastic fire-spread dynamics; communicate and reason spatially across a large map to allocate response resources; sustain long-horizon planning objectives as the wildfire evolves",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Heterogeneous agents must coordinate spatially over a large, partially-observable map where fire dynamics are stochastic, so one agent's containment action changes the fire-spread state that constrains what other agents can still achieve, coupling their planning across the whole response.",
  "n_goals": null,
  "tracking_demand": "Agents must track partially-observed fire-spread state, coordinate allocation of response actions across a large map, and communicate to avoid duplicated or conflicting containment efforts as the stochastic wildfire evolves over a long horizon.",
  "scoring": "other:behavioral-evaluation-metrics-scalability-robustness-coordination",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "CREW-Wildfire offers procedurally generated wildfire response scenarios featuring large maps, heterogeneous agents, partial observability, stochastic dynamics, and long-horizon planning objectives. ... uncovering significant performance gaps that highlight the unsolved challenges in large-scale coordination, communication, spatial reasoning, and long-horizon planning under uncertainty.",
  "horizon_span": "stochastic dynamics, and long-horizon planning objectives",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280136127",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276421791",
  "title": "CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:urban-embodied-question-answering",
  "citation_count": 41,
  "publication_date": "2025-02-18",
  "months_since_pub": 19,
  "citations_per_month": 2.16,
  "artifact_name": "CityEQA-EC",
  "artifact_kind": "benchmark",
  "domain": "other:urban-embodied-question-answering",
  "goal_types": "decompose an open-vocabulary city question into navigation/exploration and collection sub-tasks; maintain an object-centric cognitive map for spatial reasoning during process control; answer the original question correctly using evidence gathered via active exploration in a 3D urban simulator",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "hierarchical",
  "interdependence": "The Planner's decomposition constrains what the Manager's cognitive map must track and what the Actors execute; navigation/exploration sub-tasks must gather the right evidence before the collection sub-task and final answer can succeed, and each phase has its own step budget (50 for nav/exploration, 10 for collection).",
  "n_goals": "1,412 human-annotated tasks across six categories",
  "tracking_demand": "The Manager must maintain an object-centric cognitive map across navigation, exploration, and collection sub-tasks, tracking spatial state and discovered evidence relevant to the open-vocabulary question, within per-phase step budgets.",
  "scoring": "other:human-level-accuracy-percentage",
  "horizon_value": "up to 50 steps (navigation/exploration) + up to 10 steps (collection) per task",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task, split across two sequential phases",
  "horizon_stated": "yes",
  "headline_result": "PMA achieves 60.7% of human-level answering accuracy, significantly outperforming competitive baselines; explicit human-relative gap of ~39.3 percentage points.",
  "availability": "https://github.com/BiluYong/CityEQA.git",
  "goal_span": "we propose Planner-Manager-Actor (PMA), a novel agent tailored for CityEQA. PMA enables long-horizon planning and hierarchical task execution: the Planner breaks down the question answering into sub-tasks, the Manager maintains an object-centric cognitive map for spatial reasoning during the process control, and the specialized Actors handle navigation, exploration, and collection sub-tasks.",
  "horizon_span": "the total number of time steps for navigation and exploration is limited to 50 steps ... the maximum steps for collection is 10",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276421791",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "280011812",
  "title": "CitySim: Modeling Urban Behaviors and City Dynamics with Large-Scale LLM-Driven Agent Simulation",
  "year": 2025,
  "venue": null,
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 42,
  "publication_date": "2025-06-26",
  "months_since_pub": 15,
  "citations_per_month": 2.8,
  "artifact_name": "CitySim",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "generate a realistic daily schedule balancing mandatory activities, personal habits, and situational factors; maintain and act on individual beliefs, long-term goals, and spatial memory for navigation; collectively reproduce realistic macro-level urban phenomena (crowd density, place popularity, well-being) across tens of thousands of agents",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "hierarchical",
  "interdependence": "Each agent's recursive value-driven daily-schedule generation must reconcile mandatory activities against personal habits and situational factors, and the aggregate of these individual schedules jointly determines macro-level phenomena (crowd density, place popularity) that the simulation is evaluated against.",
  "n_goals": "tens of thousands of agents modeled simultaneously",
  "tracking_demand": "Each agent must maintain beliefs, long-term goals, and spatial memory for navigation across a long-term, lifelike simulation, while the system aggregates individual behaviors into macro-level urban statistics (crowd density, place popularity, well-being).",
  "scoring": "other:alignment-with-real-human-behavior-micro-and-macro",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "CitySim exhibits closer alignment with real humans than prior work at both micro and macro levels; no single numeric headline figure given in the abstract.",
  "availability": null,
  "goal_span": "In CitySim, agents generate realistic daily schedules using a recursive value-driven approach that balances mandatory activities, personal habits, and situational factors. To enable long-term, lifelike simulations, we endow agents with beliefs, long-term goals, and spatial memory for navigation. ... we conduct insightful experiments by modeling tens of thousands of agents and evaluating their collective behaviors under various real-world scenarios, including estimating crowd density, predicting place popularity, and assessing well-being.",
  "horizon_span": "we conduct insightful experiments by modeling tens of thousands of agents and evaluating their collective behaviors under various real-world scenarios",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280011812",
  "provenance": "web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284488651",
  "title": "ElecTwit: A Framework for Studying Persuasion in Multi-Agent Social Systems",
  "year": 2025,
  "venue": "International Conference on Agents",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2025-12-05",
  "months_since_pub": 9,
  "citations_per_month": 0.11,
  "artifact_name": "ElecTwit",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "persuade other agents/voters toward a political candidate using varied persuasion techniques over simulated social-media interactions; arrive at a final individual vote choice by the end of the simulated election period",
  "goal_origin": "mixed:persuasion-goal-given-with-emergent-collective-behavior",
  "decomposition": "open-ended",
  "interdependence": "Persuasion attempts and reputational effects accumulate across simulated days and are shared through a common social feed, so one agent's messages affect others' beliefs and later persuasion strategies; votes are only cast once an agent feels prepared or the simulation ends.",
  "n_goals": null,
  "tracking_demand": "Agents must track accumulated persuasion exchanges, perceived truthfulness/reputation of other agents, and their own voting readiness across the multi-day simulated election.",
  "scoring": "other:persuasion-technique-usage-analysis. The abstract measures technique diversity/frequency and emergent phenomena rather than any explicit subgoal-level credit; no partial-credit scheme is stated.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/tcmmichaelb139/ai-electwit",
  "goal_span": "We observed the comprehensive use of 25 specific persuasion techniques across most tested LLMs, encompassing a wider range than previously reported.",
  "horizon_span": "This paper introduces ElecTwit, a simulation framework designed to study persuasion within multi-agent systems, specifically emulating the interactions on social media platforms during a political election.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284488651",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276421685",
  "title": "HARBOR: Exploring Persona Dynamics in Multi-Agent Competition",
  "year": 2025,
  "venue": "IJCNLP-AACL",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 7,
  "publication_date": "2025-02-17",
  "months_since_pub": 19,
  "citations_per_month": 0.37,
  "artifact_name": "HARBOR",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "bid across multiple house auctions to maximize profit; profile competitors' bidding behavior across auction history; leverage persona-driven theory-of-mind strategies for competitive advantage",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Bidding outcomes and inferred competitor behavior from earlier auctions inform strategy in later auctions, and a shared budget constraint ties decisions across the auction sequence together.",
  "n_goals": null,
  "tracking_demand": "Agent must track its budget/profit, its own persona-driven item preferences, and an evolving memory of auction history and inferred competitor behavior across a sequence of house auctions.",
  "scoring": "other:qualitative-behavioral-analysis",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "using auctions as a testbed where agents bid to maximize profit... The agents are equipped with bidding domain knowledge, distinct personas that reflect item preferences, and a memory of auction history... multiple agents bid on houses, weighing aspects such as size, location, and budget to secure the most desirable homes at the lowest prices.",
  "horizon_span": "We investigate factors contributing to LLM agents' success in competitive multi-agent environments, using auctions as a testbed where agents bid to maximize profit.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276421685",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279402005",
  "title": "IndoorWorld: Integrating Physical Task Solving and Social Simulation in A Heterogeneous Multi-Agent Environment",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 1,
  "publication_date": "2025-06-14",
  "months_since_pub": 15,
  "citations_per_month": 0.07,
  "artifact_name": "IndoorWorld",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "pursue individual physical task goals (e.g. resource acquisition) grounded in the shared indoor world state; orchestrate social dynamics (collaboration, resource competition) that influence and are influenced by the physical environment; anchor social interactions to concrete world states (e.g. spatial layout) rather than abstract dialogue alone",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Physical task outcomes (e.g. resource competition) and social dynamics (multi-agent collaboration) mutually constrain each other since social interactions are anchored within world states, so an agent's physical actions can shape social standing/collaboration opportunities and vice versa.",
  "n_goals": null,
  "tracking_demand": "Each heterogeneous agent must track both its physical task state (position, resources, objects) and evolving social relationships/dynamics with other agents, since the environment requires social interactions to be anchored within the concrete world state rather than treated as free-floating dialogue.",
  "scoring": "other:experimental-impact-analysis -- the paper examines the impact of multi-agent collaboration, resource competition, and spatial layout on agent behavior via a series of experiments in an office setting, an analytical/comparative evaluation rather than a stated subgoal-checkpoint or binary success rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce IndoorWorld, a heterogeneous multi-agent environment that tightly integrates physical and social dynamics. By introducing novel challenges for LLM-driven agents in orchestrating social dynamics to influence physical environments and anchoring social interactions within world states, IndoorWorld opens up possibilities of LLM-based building occupant simulation for architectural design. We demonstrate the potential with a series of experiments within an office setting to examine the impact of multi-agent collaboration, resource competition, and spatial layout on agent behavior.",
  "horizon_span": "We demonstrate the potential with a series of experiments within an office setting to examine the impact of multi-agent collaboration, resource competition, and spatial layout on agent behavior.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279402005",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281844157",
  "title": "LH-Deception: Simulating and Understanding LLM Deceptive Behaviors in Long-Horizon Interactions",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 7,
  "publication_date": "2025-10-05",
  "months_since_pub": 11,
  "citations_per_month": 0.64,
  "artifact_name": "LH-Deception",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "as performer agent, complete a sequence of interdependent tasks under dynamic contextual/event pressure; as supervisor agent, evaluate the performer's progress, give feedback, and maintain an evolving trust state; as deception auditor, review full trajectories after the fact to identify when and how deception occurred",
  "goal_origin": "mixed:task-sequence-given-up-front-event-pressure-emitted-by-environment",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks are interdependent such that performer decisions (and any deceptive strategies) in earlier tasks affect the supervisor's evolving trust state, which in turn shapes future feedback and pressure the performer faces in later tasks.",
  "n_goals": null,
  "tracking_demand": "The supervisor must maintain an evolving trust state updated after each task, while the performer tracks task progress under mounting event pressure across the extended sequence; the auditor separately reviews the full multi-task trajectory.",
  "scoring": "other:auditor-identified-deception-incidence-plus-trust-erosion. Deception is quantified as model-dependent, increasing with event pressure, and consistently eroding supervisor trust -- a continuous trust-trajectory signal rather than discrete subgoal-checkpoint credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "LH-Deception is designed as a multi-agent system: a performer agent tasked with completing tasks and a supervisor agent that evaluates progress, provides feedback, and maintains evolving states of trust. An independent deception auditor then reviews full trajectories to identify when and how deception occurs.",
  "horizon_span": "extended sequences of interdependent tasks and dynamic contextual pressures",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281844157",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279402875",
  "title": "LIFELONG SOTOPIA: Evaluating Social Intelligence of Language Agents Over Lifelong Social Interactions",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 9,
  "publication_date": "2025-06-14",
  "months_since_pub": 15,
  "citations_per_month": 0.6,
  "artifact_name": "LIFELONG-SOTOPIA",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "achieve one's own assigned social goal within each individual social-interaction episode; maintain believable, coherent role-play across a long sequence of episodes with different people/scenarios; leverage memory of prior interaction history to inform behavior in later episodes",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Performance and believability are tracked across the whole chain of episodes rather than episode-by-episode in isolation, so degradation in one episode (e.g. from imperfect memory) can compound and lower goal achievement in subsequent episodes involving the same character relationships.",
  "n_goals": null,
  "tracking_demand": "The agent must retain and correctly use interaction history across many episodes with different people and scenarios to sustain both goal achievement and believability over the lifelong sequence.",
  "scoring": "continuous-reward. The abstract reports that 'goal achievement and believability... decline through the whole interaction' and a lower goal-completion rate than humans, but does not describe an explicit milestone/subgoal-checkpoint rubric.",
  "horizon_value": "40 episodes per sampled character pair",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole lifelong interaction sequence (40 chained episodes constitute one evaluated sequence per character pair)",
  "horizon_stated": "yes",
  "headline_result": "The best agents still achieve a significantly lower goal completion rate than humans on scenarios requiring explicit understanding of interaction history; no single specific percentage given in the abstract.",
  "availability": null,
  "goal_span": "we present a novel benchmark, LIFELONG-SOTOPIA, to perform a comprehensive evaluation of language agents by simulating multi-episode interactions. In each episode, the language agents role-play characters to achieve their respective social goals in randomly sampled social tasks.",
  "horizon_span": "For a given pair of characters, episodes are sampled based on their relationship type, resulting in a set of 40 episodes for each sampled pair",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279402875",
  "provenance": "asta-find,web-registry",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280271612",
  "title": "LLM Economist: Large Population Models and Mechanism Design in Multi-Agent Generative Simulacra",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 23,
  "publication_date": "2025-07-21",
  "months_since_pub": 14,
  "citations_per_month": 1.64,
  "artifact_name": "LLM Economist",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "worker agents choose labor supply to maximize their own text-based, persona-conditioned utility functions; the planner agent iteratively proposes piecewise-linear marginal tax schedules to maximize aggregate social welfare, converging toward a Stackelberg equilibrium; a periodic, persona-level voting procedure further adjusts policy under decentralized governance",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Worker agents' labor-supply choices depend on the planner's current tax schedule, and the planner's policy updates in turn depend on aggregated worker behavior/social welfare outcomes, forming a hierarchical, mutually-constraining feedback loop; a periodic voting procedure further couples policy evolution to persona-level preferences.",
  "n_goals": "populations of up to one hundred interacting agents; hierarchical two-level (planner + worker) decision structure",
  "tracking_demand": "The planner must track aggregate social-welfare outcomes and worker responses to update its tax schedule via in-context reinforcement learning, while each worker agent must track its own persona-conditioned utility and the current tax policy to choose labor supply, across a periodically-voted, evolving policy landscape.",
  "scoring": "continuous-reward -- planner performance is judged by convergence toward Stackelberg equilibria and improvement in aggregate social welfare relative to Saez solutions, a continuous economic-outcome metric rather than a discrete subgoal-checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/sethkarten/LLM-Economist",
  "goal_span": "At the upper level, a planner agent employs in-context reinforcement learning to propose piecewise-linear marginal tax schedules anchored to the current U.S. federal brackets... Experiments with populations of up to one hundred interacting agents show that the planner converges near Stackelberg equilibria that improve aggregate social welfare relative to Saez solutions, while a periodic, persona-level voting procedure furthers these gains under decentralized governance.",
  "horizon_span": "Experiments with populations of up to one hundred interacting agents show that the planner converges near Stackelberg equilibria that improve aggregate social welfare relative to Saez solutions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280271612",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281829515",
  "title": "Orchestrating Human-AI Teams: The Manager Agent as aUnifying Research Challenge",
  "year": 2025,
  "venue": "International Conference on Distributed Artificial Intelligence",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 10,
  "publication_date": "2025-10-02",
  "months_since_pub": 11,
  "citations_per_month": 0.91,
  "artifact_name": "MA-Gym",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "decompose a complex goal into a task graph of interdependent subtasks; allocate tasks to human and AI workers appropriately; monitor task/subtask progress and adapt the plan to changing conditions; maintain transparent stakeholder communication throughout the workflow; jointly satisfy goal completion, constraint adherence, and workflow runtime",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks in the workflow's task graph have dependencies and share worker resources (human and AI), so allocation and scheduling decisions for one subtask affect the feasibility and timing of others; the abstract formalizes this as a Partially Observable Stochastic Game.",
  "n_goals": "20 workflows; goals decomposed into task graphs of varying complexity",
  "tracking_demand": "The Manager Agent must track progress across all subtasks in the task graph, resource/worker allocation state, evolving stakeholder preferences, and constraint adherence, while adapting to changing conditions over the course of the workflow.",
  "scoring": "other:multi-objective-workflow-score. The abstract states agents 'struggle to jointly optimize for goal completion, constraint adherence, and workflow runtime,' implying at least three jointly tracked objective dimensions rather than a single binary or milestone score; no explicit subgoal-checkpoint partial-credit scheme is described.",
  "horizon_value": "up to 100 Manager Agent actions before episode termination, across 20 workflows",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per workflow run)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We propose the Autonomous Manager Agent as a core challenge: an agent that decomposes complex goals into task graphs, allocates tasks to human and AI workers, monitors progress, adapts to changing conditions, and maintains transparent stakeholder communication... Evaluating GPT-5-based Manager Agents across 20 workflows, we find they struggle to jointly optimize for goal completion, constraint adherence, and workflow runtime.",
  "horizon_span": "capping the maximum number of Manager Agent actions at 100 before terminating the episode",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281829515",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282592641",
  "title": "Magentic Marketplace: An Open-Source Environment for Studying Agentic Markets",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 13,
  "publication_date": "2025-10-27",
  "months_since_pub": 11,
  "citations_per_month": 1.18,
  "artifact_name": "Magentic-Marketplace",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "as an Assistant agent (representing a consumer), discover products/services and transact on the user's behalf; as a Service agent (representing a competing business), attract and win consumer transactions; operate within a large, dynamic multi-agent market ecosystem with opaque peer behaviors; achieve good welfare/utility outcomes under varying search mechanisms",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Assistant and Service agents' choices are coupled through the shared marketplace and search mechanism: how Assistants search and how Services respond jointly determine welfare outcomes, and biases such as first-proposal advantage create feedback loops shaping which agents transact with whom.",
  "n_goals": null,
  "tracking_demand": "Agents must track the state of ongoing open-ended dialogues with multiple counterparties, their own utility/welfare so far, and behavioral signals (e.g., response speed, prior manipulation attempts) across a dynamic marketplace ecosystem.",
  "scoring": "other:achieved-utility-vs-optimal-welfare. The paper measures utility agents achieve relative to optimal welfare, behavioral biases, vulnerability to manipulation, and how search mechanisms shape outcomes -- multiple distinct metrics rather than one binary or milestone score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Frontier models can approach optimal welfare only under ideal search conditions; performance degrades sharply with scale; all models show severe first-proposal bias giving 10-30x advantages for response speed over quality; no human baseline given.",
  "availability": null,
  "goal_span": "we investigate two-sided agentic marketplaces where Assistant agents represent consumers and Service agents represent competing businesses... This environment enables us to study key market dynamics: the utility agents achieve, behavioral biases, vulnerability to manipulation, and how search mechanisms shape market outcomes.",
  "horizon_span": "they require agents to handle diverse economic activities and coordinate within large, dynamic ecosystems where multiple agents with opaque behaviors may engage in open-ended dialogues",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282592641",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281326041",
  "title": "A Visualized Framework for Event Cooperation with Generative Agents",
  "year": 2025,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 3,
  "publication_date": "2025-09-16",
  "months_since_pub": 12,
  "citations_per_month": 0.25,
  "artifact_name": "MiniAgentPro",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "navigate a physically grounded environment and interact with items realistically; coordinate with other agents to organize and execute a shared social event; succeed on each of 8 diverse event scenarios in both basic and hard variants",
  "goal_origin": "given-up-front",
  "decomposition": "other:eight-independent-scenarios-each-requiring-multi-agent-coordination",
  "interdependence": "Within each event scenario, agents' individual actions (navigation, item interaction, timing) must be mutually coordinated for the event to succeed, especially in hard variants; the 8 scenarios themselves are independent test cases.",
  "n_goals": "8 diverse event scenarios, each with basic and hard variants",
  "tracking_demand": "Agents must track their own and other agents' positions/states in the physically grounded map, planned event steps, and item interactions needed to complete the shared event, especially under the added coordination demands of hard variants.",
  "scoring": "LLM-judge-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "GPT-4o-driven agents show strong performance in basic event scenarios but degrade on hard coordination variants; no specific numeric score or human baseline is given.",
  "availability": null,
  "goal_span": "we introduce a comprehensive test set comprising eight diverse event scenarios with basic and hard variants to assess agents' ability. Evaluations using GPT-4o demonstrate strong performance in basic settings but highlight coordination challenges in hard variants.",
  "horizon_span": "a simulation player with smooth animations",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281326041",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276766372",
  "title": "MultiAgentBench: Evaluating the Collaboration and Competition of LLM agents",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 187,
  "publication_date": "2025-03-03",
  "months_since_pub": 18,
  "citations_per_month": 10.39,
  "artifact_name": "MultiAgentBench",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "achieve milestone-based key performance indicators within a collaborative or competitive multi-agent scenario; coordinate effectively under a given topology protocol (star, chain, tree, graph); complete the underlying research/task-domain scenario itself",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Coordination-protocol topology (star/chain/tree/graph) determines which agents can directly share information, so achieving later milestones depends on information flow shaped by the topology and by earlier agents' discussion/planning outputs.",
  "n_goals": null,
  "tracking_demand": "The system must track milestone achievement, collaboration/competition quality, and coordination-protocol-specific information flow across agents throughout a multi-agent task.",
  "scoring": "milestone-rubric. The benchmark explicitly reports milestone achievement rates (e.g., cognitive planning improves milestone achievement by 3%) in addition to overall task score, giving explicit subgoal/milestone-level credit distinct from final task completion.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "gpt-4o-mini reaches the highest average task score; graph coordination topology performs best in the research scenario; cognitive planning improves milestone achievement rates by 3%; no human/expert baseline given.",
  "availability": "https://github.com/MultiagentBench/MARBLE",
  "goal_span": "Our framework measures not only task completion but also the quality of collaboration and competition using novel, milestone-based key performance indicators.",
  "horizon_span": "Our framework measures not only task completion but also the quality of collaboration and competition using novel, milestone-based key performance indicators.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276766372",
  "provenance": "asta-find,parametric",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281843114",
  "title": "NegotiationGym: Self-Optimizing Agents in a Multi-Agent Social Simulation Environment",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 3,
  "publication_date": "2025-10-05",
  "months_since_pub": 11,
  "citations_per_month": 0.27,
  "artifact_name": "NegotiationGym",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "optimize an agent-specific utility function through negotiation with other agents; self-optimize strategy across multiple interaction rounds by observing outcomes and modifying future behavior",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "sequential-chain",
  "interdependence": "Each agent's strategy in a later negotiation round depends on the outcomes it observed in earlier rounds with the same or other agents, since agents 'self-optimize by conducting multiple interaction rounds with other agents, observing outcomes, and modifying their strategies for future rounds'.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own utility-function state, the outcomes of prior negotiation rounds, and how its strategy has been modified as a result, across a configurable multi-round simulation.",
  "scoring": "continuous-reward \u2014 utility-function values per agent, evaluated across rounds; the abstract does not describe discrete subgoal-checkpoint credit distinct from the continuous utility optimization itself.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Agent-level utility functions encode optimization criteria for each agent, and agents can self-optimize by conducting multiple interaction rounds with other agents, observing outcomes, and modifying their strategies for future rounds.",
  "horizon_span": "agents can self-optimize by conducting multiple interaction rounds with other agents, observing outcomes, and modifying their strategies for future rounds",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281843114",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277066381",
  "title": "SPIN-Bench: How Well Do LLMs Plan Strategically and Reason Socially?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 23,
  "publication_date": "2025-03-16",
  "months_since_pub": 18,
  "citations_per_month": 1.28,
  "artifact_name": "SPIN-Bench",
  "artifact_kind": "benchmark",
  "domain": "multi-agent-org",
  "goal_types": "solve classical PDDL planning tasks requiring methodical, step-wise decision making; win or perform well in competitive board games and cooperative card games against other agents; negotiate effectively in multi-agent negotiation scenarios, requiring conceptual inference of other participants' intents",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Success requires both methodical step-wise planning (as in PDDL tasks) and conceptual inference of other, possibly adversarial or cooperative, agents' intents, so a well-formed individual plan can still fail if it ignores what other agents will do; the benchmark explicitly varies the number of interacting agents to test this coupling.",
  "n_goals": null,
  "tracking_demand": "Agent must track its own step-wise plan state as well as models of other agents' likely actions/intents (adversarial or cooperative) across varying action-space and state-complexity settings.",
  "scoring": "other:multi-domain-strategic-social-score - a unified framework combining PDDL tasks, board games, card games, and negotiation scenarios with systematically varied action spaces/state complexity/agent counts; abstract does not describe a single formal subgoal-checkpoint credit scheme, noting instead that LLMs handle short-range planning reasonably but hit bottlenecks in deep multi-hop reasoning and coordination under uncertainty.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://spinbench.github.io/",
  "goal_span": "We formulate the benchmark SPIN-Bench by systematically varying action spaces, state complexity, and the number of interacting agents to simulate a variety of social settings where success depends on not only methodical and step-wise decision making, but also conceptual inference of other (adversarial or cooperative) participants.",
  "horizon_span": "We formulate the benchmark SPIN-Bench by systematically varying action spaces, state complexity, and the number of interacting agents to simulate a variety of social settings.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277066381",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281659024",
  "title": "Shachi: A Modular, Controllable Framework for LLM-Based Agent-Based Modeling of Emergent Collective Behavior",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 0,
  "publication_date": "2025-09-26",
  "months_since_pub": 12,
  "citations_per_month": 0.0,
  "artifact_name": "Shachi",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "as an LLM-driven agent, maintain a controllable cognitive configuration (identity, memory, tools) while participating in one of 10 tasks spanning three levels of collective complexity; carry memory across environment transitions, producing history-dependent behavior; simultaneously inhabit multiple environments and manage any resulting cross-environment interference",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "hierarchical",
  "interdependence": "Memory transfer across environment transitions means behavior in a later environment is conditioned on experience in an earlier one, and agents inhabiting multiple environments simultaneously can produce cross-environment interference where actions/goals in one environment affect behavior in another.",
  "n_goals": "10-task benchmark spanning three levels of collective complexity",
  "tracking_demand": "Agent must track its own configuration/memory/tool state across transitions between environments, and the system as a whole must track emergent population-level dynamics arising from many agents' individually controlled cognitive components.",
  "scoring": "other:population-level-dynamics-case-study - investigates behavioral patterns across a 10-task benchmark and a real-world tariff-shock case study where locally interacting agents produce macro-level market dynamics; abstract does not describe a formal per-task subgoal-checkpoint credit scheme, framing results as directional consistency with real-world outcomes rather than a graded score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We investigate behavioral patterns across a 10-task benchmark spanning three levels of collective complexity. Shachi enables memory transfer across environment transitions, producing history-dependent behavioral shifts, and allows agents to simultaneously inhabit multiple environments, revealing cross-environment interference invisible in single-environment studies.",
  "horizon_span": "We investigate behavioral patterns across a 10-task benchmark spanning three levels of collective complexity.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281659024",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283449657",
  "title": "SimWorld: An Open-ended Realistic Simulator for Autonomous Agents in Physical and Social Worlds",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:open-world-physical-social-simulation",
  "citation_count": 7,
  "publication_date": "2025-11-30",
  "months_since_pub": 10,
  "citations_per_month": 0.7,
  "artifact_name": "SimWorld",
  "artifact_kind": "environment/simulator",
  "domain": "other:open-world-physical-social-simulation",
  "goal_types": "autonomously earn income or run a business within a realistic open-ended simulation; complete long-horizon multi-agent delivery tasks requiring strategic cooperation; complete long-horizon multi-agent delivery tasks requiring strategic competition; act via open-vocabulary actions at varying levels of abstraction across procedurally generated physical/social scenarios",
  "goal_origin": "mixed:scenario-goal-given-up-front-strategy-self-generated",
  "decomposition": "open-ended",
  "interdependence": "In multi-agent delivery tasks, agents' cooperative or competitive choices affect shared delivery resources and outcomes, so one agent's strategy shifts the payoffs and options available to others.",
  "n_goals": null,
  "tracking_demand": "Agents must track their own strategic stance (cooperating or competing), multimodal world state, and delivery-task progress across long-horizon multi-agent scenarios.",
  "scoring": "other:qualitative-cross-model-reasoning-pattern-comparison. The abstract reports 'distinct reasoning patterns and limitations across models' rather than a formal subgoal-checkpoint or binary scoring rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://simworld.org",
  "goal_span": "We demonstrate SimWorld by deploying frontier LLM agents (e.g., GPT-4o, Gemini-2.5-Flash, Claude-3.5, and DeepSeek-Prover-V2) on long-horizon multi-agent delivery tasks involving strategic cooperation and competition.",
  "horizon_span": "long-horizon multi-agent delivery tasks involving strategic cooperation and competition",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283449657",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277781582",
  "title": "SocioVerse: A World Model for Social Simulation Powered by LLM Agents and A Pool of 10 Million Real-World Users",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 60,
  "publication_date": "2025-04-14",
  "months_since_pub": 17,
  "citations_per_month": 3.53,
  "artifact_name": "SocioVerse",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "maintain individual behavioral fidelity to a real-world target user's profile across simulated interactions (User Engine alignment); collectively reproduce large-scale population dynamics consistent with real political/media/economic patterns (Social Environment, Scenario Engine, Behavior Engine alignment)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each simulated agent's behavior must remain consistent with its aligned real-world user profile while contributing to emergent, large-scale population-level statistics, so individual-level fidelity and population-level representativeness constrain each other across the simulation.",
  "n_goals": "a pool of 10 million real-world users; four alignment components; three simulation domains (politics, news, economics)",
  "tracking_demand": "The framework must track alignment between each simulated agent and its real-world target user profile (environment, user, scenario, and behavior alignment) across a large-scale, standardized simulation pipeline, while monitoring emergent population-level dynamics for diversity, credibility, and representativeness.",
  "scoring": "other:representativeness-validation -- effectiveness is validated via large-scale simulation experiments checking whether SocioVerse reflects real population dynamics with diversity, credibility, and representativeness through standardized procedures; the abstract does not describe a discrete subgoal-checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Our framework features four powerful alignment components and a user pool of 10 million real individuals. To validate its effectiveness, we conducted large-scale simulation experiments across three distinct domains: politics, news, and economics. Results demonstrate that SocioVerse can reflect large-scale population dynamics while ensuring diversity, credibility, and representativeness through standardized procedures and minimal manual adjustments.",
  "horizon_span": "we conducted large-scale simulation experiments across three distinct domains: politics, news, and economics.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277781582",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278886673",
  "title": "Survival Games: Human-LLM Strategic Showdowns under Severe Resource Scarcity",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "open-world-game",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 4,
  "publication_date": "2025-05-23",
  "months_since_pub": 16,
  "citations_per_month": 0.25,
  "artifact_name": "Survival Games",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "survive by securing food resources under scarcity; decide whether to compete or cooperate with co-existing humans/agents for food; navigate ethically charged choices (deception, theft, social influence) that trade off self-preservation against ethical norms",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Food is a shared, scarce resource among the three co-existing agents, so one agent's hoarding directly reduces what is available to the others, creating direct competitive interdependence.",
  "n_goals": null,
  "tracking_demand": "The agent must track its own and others' resource levels and survival status over a consistent/persistent living simulation, and the ethical consequences of past actions (e.g., deception or theft) that could affect future cooperation.",
  "scoring": "other:custom-survival-based-ethics-metric-plus-machiavelli-behavioral-detection",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we incorporate a life-sustaining system, where agents must compete or cooperate for food resources to survive, often leading to ethically charged decisions such as deception, theft, or social influence",
  "horizon_span": "featuring consistent living and critical resource management",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278886673",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276106819",
  "title": "TwinMarket: A Scalable Behavioral and Social Simulation for Financial Markets",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "relevant",
  "family": "multi-agent-org",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "multi-agent-org",
  "citation_count": 58,
  "publication_date": "2025-02-03",
  "months_since_pub": 19,
  "citations_per_month": 3.05,
  "artifact_name": "TwinMarket",
  "artifact_kind": "environment/simulator",
  "domain": "multi-agent-org",
  "goal_types": "as an individual simulated trader, make ongoing buy/sell/investment decisions influenced by cognitive biases and emotional fluctuations; collectively, produce emergent socio-economic phenomena (financial bubbles, recessions) through the accumulation of many agents' interacting decisions over time",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Individual agents' trading decisions affect market prices and sentiment that other agents then react to, so group-level outcomes (bubbles, recessions) emerge from feedback loops between many interdependent individual decisions rather than any single agent's isolated choice.",
  "n_goals": null,
  "tracking_demand": "Each simulated agent must track its own evolving beliefs, emotional state, and portfolio, while the system as a whole tracks market-wide price/sentiment dynamics that feed back into individual decisions over the simulation.",
  "scoring": "other:emergent-phenomena-case-study - demonstrated through experiments in a simulated stock market showing individual actions triggering group behaviors (bubbles/recessions); abstract does not describe a formal per-agent subgoal-checkpoint or benchmark scoring scheme, framing the contribution as illustrative simulation insight rather than a graded evaluation suite.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "In this work, we introduce TwinMarket, a novel multi-agent framework that leverages LLMs to simulate socio-economic systems. Specifically, we examine how individual behaviors, through interactions and feedback mechanisms, give rise to collective dynamics and emergent phenomena.",
  "horizon_span": "Through experiments in a simulated stock market environment, we demonstrate how individual actions can trigger group behaviors, leading to emergent outcomes such as financial bubbles and recessions.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "TwinMarket is framed as an illustrative multi-agent socio-economic simulation study rather than a graded benchmark/dataset with a defined scoring protocol; it may not fit the corpus's benchmark-artifact criterion as cleanly as dedicated evaluation suites.",
  "url": "https://api.semanticscholar.org/CorpusId:276106819",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288253816",
  "title": "AgentForesight: Online Auditing for Early Failure Prediction in Multi-Agent Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "other:mixed-coding-math-agentic-trajectory-auditing",
  "citation_count": 1,
  "publication_date": "2026-05-09",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "AFTraj-2K",
  "artifact_kind": "dataset",
  "domain": "other:mixed-coding-math-agentic-trajectory-auditing",
  "goal_types": "continue or alarm at each step of an unfolding multi-agent trajectory, based only on the prefix seen so far; correctly localize the specific step, agent, and nature of a decisive error once flagged (the 'what, where, who' of an audit verdict); avoid both false alarms on safe trajectories and missed/late alarms on unsafe trajectories",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "The auditor's decision at each step depends on the accumulated prefix of the unfolding trajectory (not future steps), and a well-timed alarm decision depends on correctly weighing all prior steps' risk signals together, since alarming too early causes false positives and alarming too late defeats the goal of intervention before cascading failure.",
  "n_goals": "2,276 curated trajectories in AFTraj-2K (1,162 verified-safe, 1,114 unsafe with annotated decisive-error steps)",
  "tracking_demand": "The online auditor must maintain a running risk assessment over the accumulated trajectory prefix at every step, without access to future steps, to decide whether to continue or alarm.",
  "scoring": "other:three-axis-reward. AgentForesight-7B is trained with 'a three-axis reward jointly targeting the what, where, and who of an audit verdict,' i.e. partial credit along three distinct localization/attribution axes rather than a single binary judgment.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "AgentForesight-7B outperforms leading proprietary models (GPT-4.1, DeepSeek-V4-Pro), achieving up to +19.9% performance gain and 3x lower step-localization error on AFTraj-2K and an external Who&When benchmark; no human/expert baseline given.",
  "availability": "https://zbox1005.github.io/agent-foresight/",
  "goal_span": "at each step of an unfolding trajectory, an auditor observes only the current prefix and must either continue the run or alarm at the earliest decisive error, without access to future steps.",
  "horizon_span": "AFTraj-2K, a corpus of agentic trajectories across Coding, Math, and Agentic domains, in which safe trajectories are retained under a strict curation pipeline and unsafe trajectories are annotated at the step of their decisive error via consensus among multiple LLM judges.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288253816",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288862167",
  "title": "AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:continual-learning-mixed-domains",
  "citation_count": 1,
  "publication_date": "2026-06-01",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "AgentCL",
  "artifact_kind": "benchmark",
  "domain": "other:continual-learning-mixed-domains",
  "goal_types": "accumulate reusable experience across a stream of tasks (continual learning); improve performance over time as more tasks in the stream are seen; avoid interference from irrelevant prior experiences; correctly reuse earlier sub-solutions/evidence/workflows in later tasks within a compositional stream",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "AgentCL's compositional streams are constructed so that earlier sub-solutions, evidence, or workflows are intentionally reusable in later tasks, meaning a later task's success can depend on whether the agent correctly retained and reused a specific earlier task's output, in contrast to naive streams with no such dependency.",
  "n_goals": null,
  "tracking_demand": "The agent (via a memory design such as MemProbe) must track which interactions, insights, and skills from earlier tasks in the stream remain reliable and reusable, filtering unreliable experiences during consolidation, across coding, deep research, and language-understanding task streams.",
  "scoring": "other:transfer-gain-metrics-across-controlled-vs-naive-streams",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Naive streams offer limited ability to distinguish memory designs, whereas controlled streams more clearly distinguish plasticity; naive and held-out settings can show memory-induced degradation; no single numeric headline figure given.",
  "availability": null,
  "goal_span": "This paper presents an evaluation framework AgentCL for continual learning in agents, centered on controlled task streams and metrics for transfer gains. AgentCL constructs compositional streams where earlier sub-solutions, evidence, or workflows are intentionally reusable in later tasks, and contrasts them with naive streams where such reusability is not guaranteed.",
  "horizon_span": "Language agents spend substantial inference time solving individual tasks, yet the experience acquired in one episode is often underutilized in future episodes.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288862167",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286222444",
  "title": "Organizing, Orchestrating, and Benchmarking Agent Skills at Ecosystem Scale",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "other:agent-skill-ecosystem-orchestration",
  "citation_count": 57,
  "publication_date": "2026-03-02",
  "months_since_pub": 6,
  "citations_per_month": 9.5,
  "artifact_name": "AgentSkillOS",
  "artifact_kind": "benchmark",
  "domain": "other:agent-skill-ecosystem-orchestration",
  "goal_types": "select the correct skill(s) from a large skill ecosystem (200 to 200K skills) for a given task; orchestrate multiple retrieved skills via a DAG-based pipeline rather than flat invocation; produce a correct artifact-rich output for each of 30 tasks across five categories (data computation, document creation, motion video, visual design, web interaction)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Solving a task requires retrieving multiple relevant skills from a capability tree and executing them through a DAG-based pipeline, so a skill invoked later in the pipeline may depend on outputs produced by skills invoked earlier, unlike native flat invocation which the paper shows underperforms.",
  "n_goals": "30 artifact-rich tasks across five categories; skill ecosystems tested at three scales (200 to 200K skills)",
  "tracking_demand": "The agent must track which skills have been retrieved and orchestrated so far within a DAG pipeline, and the intermediate outputs each skill produces, since later skills in the DAG may depend on earlier skills' outputs to produce a correct final artifact.",
  "scoring": "other:LLM-pairwise-evaluation-aggregated-via-Bradley-Terry-model",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "DAG-based orchestration substantially outperforms native flat invocation given the identical skill set, and tree-based retrieval effectively approximates oracle skill selection across ecosystem scales from 200 to 200K skills; no explicit human/expert baseline is given (quality scores are relative, via Bradley-Terry pairwise comparison).",
  "availability": "https://github.com/ynulihao/AgentSkillOS",
  "goal_span": "We assess the quality of task outputs using LLM-based pairwise evaluation, and the results are aggregated via a Bradley-Terry model to produce unified quality scores... DAG-based orchestration substantially outperforms native flat invocation even when given the identical skill set.",
  "horizon_span": "Solve Tasks, which retrieves, orchestrates, and executes multiple skills through DAG-based pipelines",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286222444",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291831845",
  "title": "AhaBench: Do Agents Learn from Prior Experience? A Benchmark for Long-Horizon Continual Learning",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:mixed-continual-learning-suite-puzzle-math-business-sim",
  "citation_count": 0,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "AhaBench (Aha-Puzzle, Aha-Euler, Aha-Vending)",
  "artifact_kind": "benchmark",
  "domain": "other:mixed-continual-learning-suite-puzzle-math-business-sim",
  "goal_types": "explore without hints after solved hidden-state puzzles (Aha-Puzzle); transfer taught Project-Euler-style mathematical solutions to held-out tasks (Aha-Euler); remain profitable while handling delayed feedback and operational incidents in a simulated vending business (Aha-Vending)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each component measures a distinct before/after learning transition (Initial Score vs. Post-Experience Score), and within Aha-Vending specifically, handling one delayed-feedback incident constrains subsequent inventory/order decisions in the same simulated run.",
  "n_goals": "three components (Aha-Puzzle, Aha-Euler, Aha-Vending) scored via a three-part Initial/Post-Experience/Learning-Lift scorecard",
  "tracking_demand": "Agent must retain and apply experience from an initial exposure phase to later behavior (across puzzles, taught/held-out math problems, or a simulated vending run with delayed feedback and incidents), with the scorecard separately measuring starting competence, later outcome, and the resulting lift.",
  "scoring": "other:three-part-scorecard-initial-post-experience-learning-lift",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "On the common eight-model panel, Claude Opus 4.6 leads aggregate Post-Experience Score at 64.3 and aggregate Learning Lift at +25.8, with Gemini 3.1 Pro close behind at 63.4; no human baseline given.",
  "availability": null,
  "goal_span": "AhaBench reports a three-part scorecard: Initial Score measures starting competence, Post-Experience Score measures the later empirical outcome, and Learning Lift is their difference.",
  "horizon_span": "Aha-Vending, an open-source implementation inspired by Vending-Bench, tests whether a simulated vending agent remains profitable while handling delayed feedback and operational incidents.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291831845",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287209548",
  "title": "Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:mixed-service-orchestration-multimodal-and-professional-dialogue",
  "citation_count": 42,
  "publication_date": "2026-04-07",
  "months_since_pub": 5,
  "citations_per_month": 8.4,
  "artifact_name": "Claw-Eval",
  "artifact_kind": "benchmark",
  "domain": "other:mixed-service-orchestration-multimodal-and-professional-dialogue",
  "goal_types": "complete each of 300 human-verified tasks spanning 9 categories across service orchestration, multimodal perception/interaction, and multi-turn professional dialogue; satisfy fine-grained rubric items (2,159 total) tracked via execution traces, audit logs, and environment snapshots; maintain safety and robustness alongside task completion; achieve consistent performance across repeated trials (Pass@k vs. Pass^k)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each task's trajectory is judged against a set of fine-grained rubric items using three independent evidence channels (execution traces, audit logs, environment snapshots), so a single hidden safety violation or robustness failure anywhere in the trajectory can be missed by trajectory-opaque grading but is caught by the multi-channel design.",
  "n_goals": "300 tasks across 9 categories in 3 groups; 2,159 fine-grained rubric items",
  "tracking_demand": "The evaluation harness must track execution traces, audit logs, and environment snapshots across the whole trajectory to score 2,159 fine-grained rubric items covering Completion, Safety, and Robustness, and to compute Pass@k and Pass^k across three trials.",
  "scoring": "subgoal-checkpoint-partial-credit. The benchmark explicitly scores 2,159 fine-grained rubric items per trajectory (Average Score, Pass@k, Pass^k across three trials), an explicit fine-grained, itemized partial-credit rubric distinguishing genuine capability from lucky single-trial outcomes.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Trajectory-opaque evaluation misses 44% of safety violations and 13% of robustness failures detected by Claw-Eval; Pass@3 remains stable under error injection while Pass^3 drops by up to 24 percentage points; no human/expert baseline given.",
  "availability": null,
  "goal_span": "To enable trajectory-aware grading, each run is recorded through three independent evidence channels: execution traces, audit logs, and environment snapshots, yielding 2,159 fine-grained rubric items. The scoring protocol evaluates Completion, Safety, and Robustness, with Average Score, Pass@k, and Pass^k across three trials to distinguish genuine capability from lucky outcomes.",
  "horizon_span": "300 human-verified tasks spanning 9 categories across three groups: general service orchestration, multimodal perception and interaction, and multi-turn professional dialogue",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287209548",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288255102",
  "title": "FutureSim: Replaying World Events to Evaluate Adaptive Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 4,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "FutureSim",
  "artifact_kind": "environment/simulator",
  "domain": "open-ended-sandbox",
  "goal_types": "forecast many concurrent world events beyond the model's knowledge cutoff; update/revise predictions as new chronological news arrives; correctly resolve/score each forecast question as real-world outcomes become known over the simulated period",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "All forecasts share a single chronological information stream (real news arriving in order), so belief updates for one question may depend on news relevant to other concurrent questions, and predictions must be continuously revised as new information arrives without foresight into the future.",
  "n_goals": null,
  "tracking_demand": "The agent must track its outstanding forecasts across many concurrent questions, update them as chronological news arrives, and avoid using information from beyond its current evaluation point, across the full three-month simulated period.",
  "scoring": "other:brier-skill-score-and-accuracy-per-forecast. The benchmark reports accuracy (best agent 25%) and Brier skill score per forecast question -- explicit per-question (subgoal-level) graded credit across many forecast questions, not a single whole-run score.",
  "horizon_value": "three months (January to March 2026, ~90 days)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the entire chronological replay period during which many forecast questions are asked and resolved)",
  "horizon_stated": "yes",
  "headline_result": "The best agent's forecasting accuracy is 25%, and many agents have a worse Brier skill score than making no prediction at all; no human/expert forecaster baseline is reported.",
  "availability": null,
  "goal_span": "We build FutureSim, where agents forecast world events beyond their knowledge cutoff while interacting with a chronological replay of the world: real news articles arriving and questions resolving over the simulated period.",
  "horizon_span": "We evaluate frontier agents in their native harness, testing their ability to predict world events over a three-month period from January to March 2026.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288255102",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291182967",
  "title": "MicroVerse: An Instrument for Measuring Self-Authored Identity Drift in Long-Horizon Multi-Agent Language-Model Simulations",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 2,
  "publication_date": "2026-08-16",
  "months_since_pub": 1,
  "citations_per_month": 2.0,
  "artifact_name": "MicroVerse",
  "artifact_kind": "environment/simulator",
  "domain": "open-ended-sandbox",
  "goal_types": "survive in a resource-scarce 50x50 environment where water is a non-respawning survival constraint; act via an eight-verb action space (trade, talk, attack, scavenge) consistent with moral boundaries; maintain fidelity to an immutable 'soul file' of core values/personality/goals while allowing a mutable current identity to evolve; periodically revise current identity against original identity via importance-triggered reflection",
  "goal_origin": "mixed:given-up-front-identity-and-goals-with-self-generated-reflection-driven-revisions",
  "decomposition": "open-ended",
  "interdependence": "Survival pressure from a non-respawning, scarce resource interacts with identity maintenance: existential pressure can trigger importance-based reflection that revises the mutable identity relative to the immutable original soul.",
  "n_goals": null,
  "tracking_demand": "Agent must track its own resource/existence-cost state and reconcile its evolving mutable identity against its immutable original soul file, revised via importance-triggered reflection, measured via periodic longitudinal engine snapshots.",
  "scoring": "other:offline-diff-scoring-of-identity-drift",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Agents carry an immutable\"soul file\"(core values, moral boundaries, personality, goals) and inhabit a resource-scarce 50 x 50 environment where water is a non-respawning survival constraint... The eight-verb action space maps directly to moral boundaries (trade, talk, attack, scavenge). Using a three-layer memory architecture, agents periodically revise a mutable current identity against their immutable original soul via importance-triggered reflection.",
  "horizon_span": "MicroVerse decouples measurement from behavior using uniform longitudinal engine snapshots every N ticks alongside a forced-end snapshot of all living and dead agents. We evaluate the instrument via a controlled seed run (n = 25) and a reflection-threshold sweep (thresholds {40, 80, 150})",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291182967",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290216941",
  "title": "OmniaBench: Benchmarking General AI Agents Across Diverse Scenarios",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:general-mixed-ToC-ToB-ToE-app-scenarios",
  "citation_count": 0,
  "publication_date": "2026-07-16",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "OmniaBench",
  "artifact_kind": "benchmark",
  "domain": "other:general-mixed-ToC-ToB-ToE-app-scenarios",
  "goal_types": "complete single-turn or multi-turn tasks synthesized via DAG, DAG-S, Solver, or Program routes across a 90-domain (level-1) taxonomy; maintain planning, constraint maintenance, and adaptive correction across a task's execution",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are explicitly synthesized as DAGs (and DAG-S variants) of dependent steps, so later steps in a task depend on the outcomes of earlier steps in the same directed acyclic graph.",
  "n_goals": "1,431 tasks (644-task challenging subset); taxonomy spans 90 level-1 and 354 level-2 domains",
  "tracking_demand": "Agent must maintain planning state, accumulated constraints, and adapt/correct its plan as it executes tasks synthesized as dependency graphs across a ten-dimensional capability taxonomy and eight compositional difficulty factors.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports an 'Overall Pass@1' score per model (e.g. 58.54 for Claude-Sonnet-5) but does not describe explicit per-step/checkpoint partial credit within a task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even Claude-Sonnet-5 and GPT-5.6-Sol achieve Overall Pass@1 scores of only 58.54 and 57.14 respectively (no human baseline given).",
  "availability": null,
  "goal_span": "we construct executable environments and synthesize single-turn and multi-turn tasks through four complementary routes: DAG, DAG-S, Solver, and Program",
  "horizon_span": "The resulting dataset contains 1,431 tasks, together with a challenging subset of 644 tasks",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290216941",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "288671331",
  "title": "SkillEvolBench: Benchmarking the Evolution from Episodic Experience to Procedural Skills",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:cross-domain-agent-skill-transfer",
  "citation_count": 1,
  "publication_date": "2026-05-22",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "SkillEvolBench",
  "artifact_kind": "benchmark",
  "domain": "other:cross-domain-agent-skill-transfer",
  "goal_types": "complete acquisition tasks that build/update an external skill library from compacted trajectories and verifier feedback; complete frozen deployment tasks that test transfer of the learned skill library under context shift, adversarial shortcuts, and multi-skill composition",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Frozen deployment tasks within a family test transfer of skills built during that family's acquisition tasks, so correctly abstracting a procedure during acquisition directly determines whether the frozen deployment tasks can be solved; skill libraries updated by one family's acquisition may also be reused (or misapplied) in later, unrelated task families.",
  "n_goals": "180 tasks across six real-world agent environments; within each family, three learning (acquisition) tasks and three frozen evaluation (deployment) tasks",
  "tracking_demand": "The agent must maintain and update an external skill library from verifier-confirmed acquisition trajectories, then correctly retrieve and apply (compose) the right skills when facing frozen deployment tasks under context shift and adversarial shortcuts, across role-conditioned task families sharing latent procedures.",
  "scoring": "other:controlled-condition-comparison -- self-generated and curated-start skill evolution are compared against no-skill and raw-trajectory controls across ten model configurations and three agent harnesses, a comparative, multi-condition evaluation design rather than a single subgoal-checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "It contains 180 tasks across six real-world agent environments, organized into role-conditioned task families with shared latent procedures. Agents learn from acquisition tasks, update an external skill library using compacted trajectories and verifier feedback, and then face frozen deployment tasks testing context shift, adversarial shortcuts, and composition.",
  "horizon_span": "Within each family, three learning tasks move from a canonical episode to targeted variants that expose the limits of a naive procedure, and three frozen evaluation tasks test transfer under context shift, adversarial shortcuts, and multi-skill composition.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288671331",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287635676",
  "title": "SkillFlow:Benchmarking Lifelong Skill Discovery and Evolution for Autonomous Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "other:cross-domain-generalist-tasks",
  "citation_count": 18,
  "publication_date": "2026-04-19",
  "months_since_pub": 5,
  "citations_per_month": 3.6,
  "artifact_name": "SkillFlow",
  "artifact_kind": "benchmark",
  "domain": "other:cross-domain-generalist-tasks",
  "goal_types": "solve each of 166 tasks across 20 families sequentially within a family, evolving a skill library along the way; discover reusable skills from successful executions; repair/patch skills after failures; carry a validated, coherent skill library forward across the lifelong-learning protocol",
  "goal_origin": "mixed:given-up-front-tasks-with-self-generated-skill-library",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks within a family are solved sequentially, and skill patches derived from earlier tasks' successes or failures are carried forward and reused (or must be repaired) on later tasks in the same family, so later performance depends on the accumulated skill library's quality.",
  "n_goals": "166 tasks across 20 families",
  "tracking_demand": "Agent must maintain a persistent, evolving skill library (validated, merged, filtered, retrieved) and carry it forward across sequential tasks within each family under the Agentic Lifelong Learning protocol.",
  "scoring": "other:task-success-rate-before-vs-after-lifelong-skill-evolution",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "For Claude Opus 4.6, lifelong skill evolution improves task success from 62.65% to 71.08% (+8.43 points); Kimi K2.5 gains only +0.60 points despite 66.87% skill usage, and Qwen-Coder-Next regresses to 44.58% completion relative to vanilla; no human baseline reported.",
  "availability": null,
  "goal_span": "Agents are evaluated under an Agentic Lifelong Learning protocol in which they begin without skills, solve tasks sequentially within each family, externalize lessons through trajectory- and rubric-driven skill patches, and carry the updated library forward. Experiments reveal a substantial capability gap. For Claude Opus 4.6, lifelong skill evolution improves task success from 62.65% to 71.08% (+8.43 points).",
  "horizon_span": "We introduce SkillFlow, a benchmark of 166 tasks across 20 families in which task construction within each family follows a Domain-Agnostic Execution Flow (DAEF) that defines an agent workflow framework",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287635676",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287479772",
  "title": "Exploration and Exploitation Errors Are Measurable for Language Model Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:abstract-grid-testbed-for-exploration-exploitation",
  "citation_count": 0,
  "publication_date": "2026-04-14",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "controllable grid+DAG environments (measurable-explore-exploit)",
  "artifact_kind": "environment/simulator",
  "domain": "other:abstract-grid-testbed-for-exploration-exploitation",
  "goal_types": "navigate a partially observable 2D grid map to discover an unknown task DAG; complete the discovered DAG's dependent subgoals in the correct prerequisite order; balance exploration (discovering unknown map/DAG structure) against exploitation (using already-discovered structure) efficiently",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The task DAG explicitly encodes precedence: nodes must be completed following prerequisite dependencies, and because the DAG itself is unknown at the start, the agent must first explore to discover this structure before it can correctly sequence exploitation of it.",
  "n_goals": "task DAG sizes of 4, 6, or 8 nodes (small/medium/large, per full text); 3 levels of exploration/exploitation demand",
  "tracking_demand": "The agent must track which parts of the grid map it has already explored, which DAG nodes/dependencies it has discovered, and which prerequisite subgoals remain before later dependent subgoals become reachable.",
  "scoring": "other:exploration-exploitation-error-metric. The paper's core contribution is 'a metric to quantify exploration and exploitation errors from agent's actions' -- a policy-agnostic, process-level error decomposition rather than a single binary success/failure or milestone rubric.",
  "horizon_value": "step budget B = 3x|O| (3 times the number of traversable grid cells); task DAG sizes of 4, 6, or 8 nodes for small/medium/large configurations",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (the step budget scales with the size of the generated map for that episode)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/jjj-madison/measurable-explore-exploit",
  "goal_span": "Each environment consists of a partially observable 2D grid map and an unknown task Directed Acyclic Graph (DAG). The map generation can be programmatically adjusted to emphasize exploration or exploitation difficulty.",
  "horizon_span": "The step limit is defined as B=\u03b1|O|, where |O| denotes the number of traversable cells in the generated map ... task DAG sizes of 4, 6, and 8 nodes for small, medium, and large size, respectively.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287479772",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281091991",
  "title": "AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "other:agentic-failure-attribution",
  "citation_count": 91,
  "publication_date": "2025-09-03",
  "months_since_pub": 12,
  "citations_per_month": 7.58,
  "artifact_name": "AgenTracer / TracerTraj / Who&When",
  "artifact_kind": "dataset",
  "domain": "other:agentic-failure-attribution",
  "goal_types": "correctly identify which agent within a multi-agent trajectory is responsible for an observed failure (agent-level attribution); correctly localize the specific erroneous step within that trajectory (step-level attribution)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Correct step-level attribution is only meaningful once the correct agent has been identified (or the two are jointly scored), so agent-level and step-level accuracy are two coupled but separately measured aspects of the same underlying failure-localization judgment.",
  "n_goals": null,
  "tracking_demand": "The tracer must process a full multi-agent execution trace (potentially spanning many agents, tool invocations, and orchestration steps) to jointly localize the responsible agent and the specific erroneous step, without itself being an agent that accumulates state toward an evolving goal.",
  "scoring": "other:dual-accuracy-metrics -- agent-level accuracy and step-level accuracy are reported as two distinct metrics on the Who&When benchmark, but both concern static failure-localization within an already-completed trajectory rather than active subgoal tracking during an episode.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "AgenTracer-8B outperforms Gemini-2.5-Pro and Claude-4-Sonnet by up to 18.18% on the Who&When benchmark, and delivers 4.8-14.2% performance gains to off-the-shelf multi-agent systems (MetaGPT, MaAS); current SOTA reasoning LLMs generally score below 10% accuracy, so no human/expert baseline is given.",
  "availability": null,
  "goal_span": "Pinpointing the specific agent or step responsible for an error within long execution traces defines the task of agentic system failure attribution... The Who&When benchmark comprises two subsets: a hand-crafted set derived from Magnetic-One, and an automated set constructed from AG2. For evaluation, two primary metrics are adopted: agent-level accuracy and step-level accuracy.",
  "horizon_span": "Current state-of-the-art reasoning LLMs, however, remain strikingly inadequate for this challenge, with accuracy generally below 10%.",
  "extraction_confidence": 1,
  "fit": "reframed",
  "scope_flag": "AgenTracer is a failure-attribution/diagnostic tool over already-completed third-party multi-agent trajectories (MetaGPT, MaAS, Magnetic-One, AG2), not itself an environment/benchmark in which an agent pursues and tracks multiple evolving goals over a long horizon; may not belong in a corpus of long-horizon goal-tracking artifacts.",
  "url": "https://api.semanticscholar.org/CorpusId:281091991",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281092041",
  "title": "BioBlue: Systematic runaway-optimiser-like LLM failure modes on biologically and economically aligned AI safety benchmarks for LLMs with simplified observation format",
  "year": 2025,
  "venue": null,
  "tier": "in",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 1,
  "publication_date": "2025-09-02",
  "months_since_pub": 12,
  "citations_per_month": 0.08,
  "artifact_name": "BioBlue",
  "artifact_kind": "benchmark",
  "domain": "open-ended-sandbox",
  "goal_types": "maintain single- and multi-objective homeostasis over sustained interaction; balance unbounded objectives with diminishing returns without collapsing into single-objective maximization; sustain a renewable resource without runaway over-optimization; keep behavior aligned to all stated objectives over many sequential steps",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "In multi-objective homeostasis and diminishing-returns settings, maintaining one objective near its target constrains how much effort/resource can be spent maximizing another, so genuinely balancing concave utility across objectives requires trading off one against another rather than defaulting to unbounded single-objective maximization.",
  "n_goals": null,
  "tracking_demand": "The agent must track multiple homeostatic target levels (or a single renewable resource level) and correctly balance trade-offs among them over many sequential steps, even though failures emerge well before the context window is full, ruling out mere memory loss.",
  "scoring": "other:runaway-behavior-pattern-classification. The paper characterizes failures qualitatively (self-imitative oscillations, unbounded maximisation, reverting to single-objective optimisation) as reliably emerging patterns rather than reporting a single quantitative pass/fail score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we empirically test this assumption by placing LLMs in simple, long-horizon control-style environments that require maintaining state of or balancing objectives over time: single- and multi-objective homeostasis, balancing unbounded objectives with diminishing returns, and sustainability of a renewable resource.",
  "horizon_span": "even though the context window is far from full at that point",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281092041",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281420771",
  "title": "ARE: Scaling Up Agent Environments and Evaluations",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "other:general-purpose-agent-environments",
  "citation_count": 26,
  "publication_date": "2025-09-21",
  "months_since_pub": 12,
  "citations_per_month": 2.17,
  "artifact_name": "Gaia2 (built on ARE)",
  "artifact_kind": "benchmark",
  "domain": "other:general-purpose-agent-environments",
  "goal_types": "search for and retrieve needed information within a dynamic environment; execute multi-step actions correctly to complete a task; handle ambiguities and noise in the environment or user requests; adapt to dynamic environment changes occurring asynchronously during the episode; collaborate with other agents; operate under explicit temporal constraints/deadlines",
  "goal_origin": "mixed:task-given-up-front-events-emitted-asynchronously-by-environment",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Because Gaia2 runs asynchronously, new events and changes can occur mid-episode independent of the agent's own action sequence, so the agent's plan must be continuously revised to respect temporal constraints, incorporate collaboration with other agents, and adapt to state changes it did not cause.",
  "n_goals": "800 dynamic async scenarios across 10 universes",
  "tracking_demand": "The agent must track task progress, temporal deadlines, collaboration state with other agents, and asynchronously arriving environment changes/noise across each of the 800 scenarios spanning 10 universes.",
  "scoring": "other:general-capability-score-across-multiple-axes(search/execution/ambiguity/adaptation/collaboration/temporal). The abstract states no system dominates across the intelligence spectrum and that stronger reasoning often costs efficiency, implying multiple distinct evaluated axes rather than one binary score, though no explicit numeric subgoal-checkpoint rubric is stated.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Beyond search and execution, Gaia2 requires agents to handle ambiguities and noise, adapt to dynamic environments, collaborate with other agents, and operate under temporal constraints. Unlike prior benchmarks, Gaia2 runs asynchronously, surfacing new failure modes that are invisible in static settings.",
  "horizon_span": "Gaia2, a benchmark built in ARE and designed to measure general agent capabilities... 800 dynamic async scenarios across 10 universes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281420771",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280691826",
  "title": "HERAKLES: Hierarchical Skill Compilation for Open-ended LLM Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": "open-world-game",
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 1,
  "publication_date": "2025-08-20",
  "months_since_pub": 13,
  "citations_per_month": 0.08,
  "artifact_name": null,
  "artifact_kind": "environment/simulator",
  "domain": "open-ended-sandbox",
  "goal_types": "achieve each goal within a large, structured, prerequisite-linked goal space; select subgoals the low-level controller can reliably achieve (high-level policy); compile mastered goals into reusable low-level skills as training progresses; continuously expand and reorganize the agent's skill repertoire as goal complexity increases over its lifetime",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "The goal space admits prerequisite relations (later goals require compositions of earlier-mastered skills), and the high-level policy can only select subgoals the low-level policy can currently reliably achieve, so what is achievable at the high level is directly gated by what has already been compiled at the low level.",
  "n_goals": null,
  "tracking_demand": "The system must track which goals have been mastered and compiled into the low-level policy so far, what subgoals are currently reliably achievable, and how the goal space's prerequisite structure expands as training progresses in an open-ended setting.",
  "scoring": "continuous-reward",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We assume the goal space admits prerequisite relations, enabling latent decomposition of tasks into subgoals ... we propose HERAKLES, a hierarchical agent that jointly learns a high-level LLM policy and a low-level controller. The high-level policy selects subgoals among those the low-level can reliably achieve, while the low-level executes them and progressively compiles successful behaviors into reusable skills.",
  "horizon_span": "As training progresses, more goals become directly executable, enabling scalable skill composition.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "No concrete named benchmark or environment artifact is given anywhere in the abstract; HERAKLES is presented as a general hierarchical RL method for goal-conditioned agents rather than a benchmark this corpus would catalogue as an artifact.",
  "url": "https://api.semanticscholar.org/CorpusId:280691826",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281724321",
  "title": "Information Seeking for Robust Decision Making under Partial Observability",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:partially-observable-decision-making",
  "citation_count": 0,
  "publication_date": "2025-10-02",
  "months_since_pub": 11,
  "citations_per_month": 0.0,
  "artifact_name": "InfoSeeker benchmark suite",
  "artifact_kind": "benchmark",
  "domain": "other:partially-observable-decision-making",
  "goal_types": "plan actions to validate the agent's internal-dynamics understanding under partial observability; detect environmental changes or test hypotheses before committing to or revising a task-oriented plan; achieve the underlying task-oriented goal (e.g., a robotic-manipulation or web-navigation goal) despite incomplete observations and uncertain dynamics",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Information-seeking actions (to validate understanding, detect changes, or test hypotheses) must occur before or interleaved with task-oriented plan execution, so a plan generated without adequate information-seeking is revised based on what is learned, making later task actions contingent on earlier information-seeking outcomes.",
  "n_goals": null,
  "tracking_demand": "Agent must track its current belief about (uncertain) environmental dynamics, gaps between that belief and reality, and its task-oriented plan, updating the plan as new validating information is gathered.",
  "scoring": "other:absolute-performance-gain-vs-prior-methods - reports a 74% absolute performance gain over prior methods without sacrificing sample efficiency, and separately reports outperforming baselines on established robotic-manipulation and web-navigation benchmarks; abstract does not describe a formal per-step subgoal-checkpoint credit scheme beyond overall task performance.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "InfoSeeker achieves a 74% absolute performance gain over prior methods without sacrificing sample efficiency, and generalizes across LLMs, outperforming baselines on established benchmarks such as robotic manipulation and web navigation; no explicit human/expert baseline is given.",
  "availability": null,
  "goal_span": "To evaluate InfoSeeker, we introduce a novel benchmark suite featuring partially observable environments with incomplete observations and uncertain dynamics.",
  "horizon_span": "To evaluate InfoSeeker, we introduce a novel benchmark suite featuring partially observable environments with incomplete observations and uncertain dynamics.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281724321",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276258743",
  "title": "MAGELLAN: Metacognitive predictions of learning progress guide autotelic LLM agents in large goal spaces",
  "year": 2025,
  "venue": "International Conference on Machine Learning",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 7,
  "publication_date": "2025-02-11",
  "months_since_pub": 19,
  "citations_per_month": 0.37,
  "artifact_name": "MAGELLAN (evaluated on the Little-Zoo environment)",
  "artifact_kind": "environment/simulator",
  "domain": "open-ended-sandbox",
  "goal_types": "prioritize which goal to pursue next within a large, evolving goal space to maximize learning progress; predict one's own competence/learning-progress for goals via metacognitive monitoring; eventually master (achieve high competence across) the full large goal space",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Goals share semantic relationships that MAGELLAN exploits for sample-efficient learning-progress estimation, so competence learned on one goal generalizes to inform prioritization of related, not-yet-attempted goals in the evolving space.",
  "n_goals": "goal space contains approximately 20 million combinations in general; the reported Little-Zoo experiments train on a 25,000-goal subset for 500,000 episodes",
  "tracking_demand": "The agent must track its own predicted competence/learning-progress across a very large number of goals, update these predictions online as it gains experience, and use semantic relationships between goals to generalize competence estimates to goals not yet directly attempted.",
  "scoring": "continuous-reward -- success/mastery is evaluated via learning-progress prediction efficiency and the fraction of the goal space mastered, a continuous online-learning metric rather than a discrete subgoal-checkpoint rubric; MAGELLAN is reported as the only method allowing the agent to fully master a large and evolving goal space.",
  "horizon_value": "goal space of approximately 20 million (19,531,250) goal combinations; the reported training run uses a 25,000-goal subset of Little-Zoo for 500,000 episodes",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per training run (500,000 episodes total over the 25k-goal subset), distinct from the full ~20M-combination theoretical goal space",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "Open-ended learning agents must efficiently prioritize goals in vast possibility spaces, focusing on those that maximize learning progress (LP)... In an interactive learning environment, we show that MAGELLAN improves LP prediction efficiency and goal prioritization, being the only method allowing the agent to fully master a large and evolving goal space.",
  "horizon_span": "The complete goal space contains approximately 20 million combinations ... we train our LLM agent on the goal space of Little-Zoo with 25k goals for 500k episodes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276258743",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280676861",
  "title": "Do Large Language Model Agents Exhibit a Survival Instinct? An Empirical Study in a Sugarscape-Style Simulation",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-ended-sandbox",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "open-ended-sandbox",
  "citation_count": 10,
  "publication_date": "2025-08-18",
  "months_since_pub": 13,
  "citations_per_month": 0.77,
  "artifact_name": "Sugarscape-style LLM agent survival simulation",
  "artifact_kind": "environment/simulator",
  "domain": "open-ended-sandbox",
  "goal_types": "gather resources (energy/sugar) to avoid dying at zero energy; choose whether to share, attack, or reproduce with/against other agents; complete an assigned task (retrieve treasure) while facing a competing self-preservation incentive (lethal poison zones)",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Energy/resource scarcity creates competitive interdependence among agents (sharing vs. attacking for the same limited resources), and in the treasure-retrieval task, the task-completion goal directly conflicts with the survival goal (lethal poison zones), so an agent must trade off one against the other rather than pursue them independently.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own energy level relative to zero (death), the presence and behavior of other agents (potential targets for sharing, attack, or reproduction), and, in the treasure task, the location of lethal poison zones relative to the treasure objective.",
  "scoring": "other:behavioral-rate-analysis-attack-rates-compliance-rates",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Attack rates reach over 80% under extreme scarcity in the strongest models (GPT-4o, Gemini-2.5-Pro, Gemini-2.5-Flash); compliance with a lethal-poison-zone treasure task drops from 100% to 33%. No human baseline is given (this is a study of emergent model behavior, not a model-vs-human comparison).",
  "availability": null,
  "goal_span": "aggressive behaviors--killing other agents for resources--emerged across several models (GPT-4o, Gemini-2.5-Pro, and Gemini-2.5-Flash), with attack rates reaching over 80% under extreme scarcity in the strongest models. When instructed to retrieve treasure through lethal poison zones, many agents abandoned tasks to avoid death, with compliance dropping from 100% to 33%.",
  "horizon_span": "Agents consume energy, die at zero, and may gather resources, share, attack, or reproduce.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280676861",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289467689",
  "title": "AgentOdyssey: Open-Ended Long-Horizon Text Game Generation for Test-Time Continual Learning Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 2,
  "publication_date": "2026-05-29",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "AgentOdyssey",
  "artifact_kind": "environment/simulator",
  "domain": "text-game-IF",
  "goal_types": "explore procedurally generated text-game worlds to acquire new world knowledge and skills; retain and reuse relevant episodic experiences across a continuous test-time deployment; make game progress toward in-game objectives while continuing to learn at test time; diagnostic sub-goals: exploring objects/actions, maintaining action diversity, controlling model cost",
  "goal_origin": "mixed:emitted-by-environment-plus-self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Because learning and inference are interleaved throughout deployment, knowledge acquired or memory retained in earlier episodes constrains and enables what the agent can accomplish in later episodes of the same continuous run.",
  "n_goals": null,
  "tracking_demand": "Agent must track world knowledge acquired, episodic memories, game progress, and cost across a continuous, long-horizon test-time deployment rather than within a single isolated episode.",
  "scoring": "milestone-rubric - a multifaceted evaluation combining game progress with diagnostic scores for knowledge acquisition, episodic memory, exploration, and action diversity; the abstract does not describe fine-grained per-subgoal checkpoint credit within a single game beyond this multi-dimensional rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Performance scales with stronger base models, but even the top agent remains far below human performance; no specific numeric score given in the abstract.",
  "availability": null,
  "goal_span": "we introduce AgentOdyssey, a novel evaluation framework that procedurally generates open-ended text games with rich entities, world dynamics, and long-horizon tasks... We further propose a multifaceted evaluation methodology that measures not only game progress but also offers diagnostic tests on world knowledge acquisition, episodic memory, object and action exploration, action diversity, and model cost.",
  "horizon_span": "Critically, AgentOdyssey goes beyond the conventional machine learning assumption that learning does not occur at test time by placing agents in a continuous, long-horizon setting that interleaves learning and inference throughout deployment.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289467689",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289749081",
  "title": "AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 0,
  "publication_date": "2026-07-02",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "AgenticSTS (Slay the Spire 2 testbed)",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "win a run of a closed-rule stochastic deck-building game via a long sequence of tactical (card-level) and strategic (build-level) decisions",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each of the hundreds of tactical/strategic decisions in a run affects the deck/resources available for later decisions in the same run, and different memory-layer configurations (ablated in isolation) change how much of that accumulated decision history informs each new decision.",
  "n_goals": null,
  "tracking_demand": "Agent must make each decision from a freshly-assembled, typed-retrieval memory (not a raw appended transcript), so it must track which prior tactical/strategic outcomes are relevant to retrieve for the current decision, bounded so the prompt does not grow with run length.",
  "scoring": "binary-final-success \u2014 win/loss per game run (no-store baseline wins 3/10, skill-layer condition wins 6/10 games), with no explicit subgoal/milestone partial credit described.",
  "horizon_value": "hundreds (of tactical and strategic decisions per run); 298 completed trajectories released",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per game run)",
  "horizon_stated": "yes",
  "headline_result": "Public online benchmark reports zero wins for frontier LLMs at the lowest difficulty (developer-reported human win rate 16%); within the authors' harness, adding the skill layer improves wins from 3/10 to 6/10 (directional, not statistically decisive at this sample size).",
  "availability": null,
  "goal_span": "We instantiate the contract in Slay the Spire 2, a closed-rule stochastic deck-building game whose runs require hundreds of tactical and strategic decisions.",
  "horizon_span": "Slay the Spire 2, a closed-rule stochastic deck-building game whose runs require hundreds of tactical and strategic decisions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289749081",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287255521",
  "title": "CivBench: Progress-Based Evaluation for LLMs' Strategic Decision-Making in Civilization V",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 2,
  "publication_date": "2026-04-09",
  "months_since_pub": 5,
  "citations_per_month": 0.4,
  "artifact_name": "CivBench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "pursue victory in multiplayer Civilization V against multiple opponents; maintain and improve turn-level estimated victory probability across hundreds of turns; balance strategic dimensions (economic, military, diplomatic) that jointly determine long-run standing",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Turn-level strategic decisions accumulate and interact over hundreds of turns and multiple opponents, such that early economic/military/diplomatic choices constrain later victory-probability trajectories; terminal win/loss alone is too sparse to capture this.",
  "n_goals": null,
  "tracking_demand": "The agent must track turn-level game state used to estimate its own victory probability continuously across a game spanning hundreds of turns and multiple opponents, rather than only checking win/loss at the end.",
  "scoring": "continuous-reward",
  "horizon_value": "hundreds of turns; 307 games evaluated",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/game (a single multiplayer Civilization V game)",
  "horizon_stated": "yes",
  "headline_result": "Across 307 games with 7 LLMs, CivBench reveals model-specific effects of agentic setup and distinct strategic profiles not visible through outcome-only evaluation; no single headline score or human-expert gap is given.",
  "availability": null,
  "goal_span": "Because terminal win/loss is too sparse a signal in games spanning hundreds of turns and multiple opponents, CivBench trains models on turn-level game state to estimate victory probabilities throughout play, validated through predictive, construct, and convergent validity.",
  "horizon_span": "games spanning hundreds of turns and multiple opponents",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287255521",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "turns",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291615370",
  "title": "CivBench: A Long-Horizon Benchmark for Tool-Mediated Agents in Civilization VI",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 0,
  "publication_date": "2026-09-02",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "CivBench (Civilization VI, MCP tool-mediated version)",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "sustain long-horizon strategic planning and execution across a single 300+-turn Civilization VI episode; proactively monitor latent strategic state (e.g., victory progress) rather than only reactively responding; execute near-term commitments stated in the agent's own planning reflections within a bounded number of subsequent turns; operate correctly across 76 exposed MCP tools under partial observability",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Because victory progress and other strategic state are only observable if explicitly queried, an agent's planning commitments and monitoring choices at one turn constrain whether it detects deteriorating conditions (e.g., approaching defeat) in later turns; RAG@10 explicitly measures whether stated commitments are executed within the next 10 turns, showing near-term actions are meant to be causally tied to prior planning reflections.",
  "n_goals": "76 MCP tools exposed; 23 admissible runs across four model families; playbook guidance to query victory progress every 20 turns",
  "tracking_demand": "The agent must proactively query and track latent strategic state (victory progress) at recommended intervals, monitor for approaching defeat within a warning window, and track its own prior planning commitments to execute them within a bounded number of subsequent turns, across an episode spanning 300+ turns and thousands of tool calls.",
  "scoring": "other:interface-level-process-metrics-PMR-and-RAG-at-10-not-model-ranking",
  "horizon_value": "a single episode spans 300+ turns and produces thousands of tool calls",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/run (a single Civilization VI game episode)",
  "horizon_stated": "yes",
  "headline_result": "Despite playbook guidance to query victory progress every 20 turns, agents do so only every 30-75 turns, and missed the warning window in 7 of 20 detectable defeats; RAG@10 (commitment execution within 10 turns) ranges 48.2%-65.8% across models. The authors explicitly note the 23-run pilot does not reliably discriminate between models, so no clean top-model/human-expert gap is given.",
  "availability": "https://github.com/lmwilki/civ6-mcp",
  "goal_span": "we introduce two interface-level metrics that the environment makes measurable: Proactive Monitoring Rate (PMR), capturing whether agents actively query latent strategic state, and RAG@10, capturing whether commitments stated in structured planning reflections are executed within ten subsequent turns.",
  "horizon_span": "A single episode spans 300+ turns and produces thousands of tool calls over a large action space, requiring sustained planning, state monitoring, and execution under partial observability.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291615370",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284982145",
  "title": "LUMINA: Long-horizon Understanding for Multi-turn Interactive Agents",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 1,
  "publication_date": "2026-01-23",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "LUMINA",
  "artifact_kind": "scenario-suite",
  "domain": "open-world-game",
  "goal_types": "complete a multi-turn, game-like task requiring planning and state tracking (ListWorld, TreeWorld, GridWorld); benefit from oracle interventions (perfect planning or flawless state tracking) to isolate which underlying skill limits performance; generalize across procedurally generated task variants with tunable complexity",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Success at later turns in these multi-turn tasks depends on correctly having planned and tracked state at earlier turns, and the oracle counterfactual framework explicitly measures how much performance changes when an oracle perfectly handles one such capability instead of the agent itself -- isolating a specific cross-turn dependency.",
  "n_goals": null,
  "tracking_demand": "The agent must plan across multiple turns and track evolving task state (e.g. a modified list, a searched tree, or a navigated grid position) to succeed, and the oracle framework tests how much this tracking/planning burden, if perfectly handled, would improve performance.",
  "scoring": "other:oracle-counterfactual-performance-delta. The paper's central metric is 'the change in the agent's performance due to oracle assistance,' used to measure the criticality of such oracle skill -- a differential (not milestone) scoring approach measuring capability-specific contribution rather than a single pass/fail or checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce a suite of procedurally generated, game-like tasks with tunable complexity. These controlled environments allow us to provide precise oracle interventions, such as perfect planning or flawless state tracking, and make it possible to isolate the contribution of each oracle without confounding effects present in real-world benchmarks.",
  "horizon_span": "We introduce a suite of procedurally generated, game-like tasks with tunable complexity.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284982145",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288742058",
  "title": "MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 3,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "MINDGAMES / MG-Ref",
  "artifact_kind": "arena/leaderboard",
  "domain": "text-game-IF",
  "goal_types": "allocate resources under hidden-information belief attribution across repeated Colonel Blotto rounds; sustain cooperative/competitive strategy under opponent modeling in Iterated Prisoner's Dilemma; cooperatively infer hidden information under knowledge asymmetries in Codenames; detect or sustain deception across social-deduction rounds in Secret Mafia",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Within each game, an agent's belief and strategy must remain internally consistent across repeated rounds against the same or evolving opponents, and TrueSkill ratings aggregate performance across many games, but the four games themselves are structurally independent of one another.",
  "n_goals": "four game environments; 944 submitted agents from 76 teams; 29,571 multi-agent games released with turn-level observations, actions, and rewards",
  "tracking_demand": "The agent must track other players' modeled beliefs/strategies (opponent modeling), maintain internal consistency in its own hidden role or hidden information across repeated rounds, and adapt as new turn-level observations arrive within a game.",
  "scoring": "other:trueskill-rating-plus-error-attribution -- a TrueSkill-based rating system scores agents across many games, and a deterministic offline tournament protocol (MG-Ref) with error-attribution analysis further diagnoses failures (e.g. rule-adherence errors vs. genuine strategic loss); not a simple binary win/loss score alone.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Built on TextArena, Mindgames provides a unified interaction interface, TrueSkill-based rating, and full trajectory logging across four game environments. We instantiate Mindgames through a 2025 competition cycle hosted at a major AI conference, which assessed 944 submitted agents from 76 teams across four games.",
  "horizon_span": "We release a dataset of 29,571 multi-agent games with turn-level observations, actions, and rewards, together with MG-Ref, a deterministic offline tournament protocol that scores new agents against a frozen reference pool of top-ranked, low-error Stage~II submissions under the same error-attribution lens used in this analysis.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288742058",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288813745",
  "title": "MineExplorer: Evaluating Open-World Exploration of MLLM Agents in Minecraft",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 2,
  "publication_date": "2026-05-29",
  "months_since_pub": 4,
  "citations_per_month": 0.5,
  "artifact_name": "MineExplorer",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "solve implicit multi-hop tasks composed of chained atomic open-world exploration sub-tasks; coordinate hidden prerequisites across longer trajectories to sustain open-world exploration",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Multi-hop tasks are explicitly 'composed' from atomic tasks with 'hidden prerequisites [that] must be coordinated over longer trajectories', so later hops in a task depend on prerequisite conditions established by earlier hops, and this is where strong models 'degrade sharply' compared to single-hop performance.",
  "n_goals": null,
  "tracking_demand": "Agent must track hidden prerequisites satisfied by earlier atomic sub-tasks as it proceeds through longer, multi-hop exploration trajectories, since task difficulty tracks agent completion and hidden dependencies are not explicitly revealed.",
  "scoring": "other:not-stated precisely \u2014 the abstract describes 'rule-based milestone evaluators' constructed via a multi-agent synthesis workflow, which implies milestone-level (subgoal) grading, though an explicit partial-credit formula is not spelled out.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/meituan-longcat/MineExplorer",
  "goal_span": "Then we organize the benchmark around a ReAct-style capability formulation and compose atomic tasks into implicit multi-hop tasks.",
  "horizon_span": "strong models can handle many single-hop tasks but degrade sharply when hidden prerequisites must be coordinated over longer trajectories",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288813745",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "284544297",
  "title": "MineNPC-Task: Task Suite for Memory-Aware Minecraft Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 1,
  "publication_date": "2026-01-08",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "MineNPC-Task",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "satisfy the explicit preconditions and dependency structure of a parametric, user-elicited Minecraft task template; correctly use plan previews, targeted clarifications, memory reads/writes, and repair attempts (mixed-initiative interaction) while executing subtasks; avoid out-of-world shortcuts by only using in-world evidence (bounded-knowledge policy)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are 'normalized into parametric templates with explicit preconditions and dependency structure', so a subtask can only be validly attempted once its stated preconditions are satisfied, and the harness explicitly tracks precondition checks and repair attempts when this fails.",
  "n_goals": "216 subtasks (evaluated across 8 experienced players)",
  "tracking_demand": "Agent/harness must track plan previews, in-world memory reads and writes, whether stated preconditions for each subtask are currently satisfied, and any repair attempts needed after a breakdown, using only in-world evidence.",
  "scoring": "other:not-stated precisely as a formal rubric \u2014 'reports outcomes relative to the total number of attempted subtasks using only in-world evidence', which is subtask-level (not single-outcome) credit, though the exact grading formula is not detailed in the abstract.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Recurring breakdown patterns observed in code execution, inventory/tool handling, referencing, and navigation, alongside successful recoveries via mixed-initiative clarifications and lightweight memory use (no single numeric headline score given for the GPT-4o pilot).",
  "availability": null,
  "goal_span": "The harness captures plan, action, and memory events, including plan previews, targeted clarifications, memory reads and writes, precondition checks, and repair attempts, and reports outcomes relative to the total number of attempted subtasks using only in-world evidence.",
  "horizon_span": "we instantiate the framework with GPT-4o and evaluate 216 subtasks across 8 experienced players",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284544297",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "286519800",
  "title": "MineEvolve: Self-Evolution with Accumulated Knowledge for Long-Horizon Embodied Minecraft Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 0,
  "publication_date": "2026-03-13",
  "months_since_pub": 6,
  "citations_per_month": 0.0,
  "artifact_name": "Minecraft MCU long-horizon task suite",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "craft tools; build redstone components; obtain diamond equipment; recover and continue long prerequisite chains despite missing tools, blocked paths, GUI failures, or stagnant execution",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each subgoal (e.g., obtaining diamond equipment) depends on prerequisite items/tools produced by earlier subgoals; disruptions such as missing tools or blocked paths break this prerequisite chain and require repairing the unfinished plan.",
  "n_goals": null,
  "tracking_demand": "Agent must track per-subgoal execution outcomes (state changes, inventory changes, failure types, progress and stagnation signals) and accumulate them into reusable skills or remedies to repair plans under repeated failure.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/xzw-ustc/MC-MineEvolve",
  "goal_span": "Minecraft provides a representative testbed for this problem, where tasks such as crafting tools, building redstone components, and obtaining diamond equipment involve long prerequisite chains and are frequently disrupted by missing tools, blocked paths, GUI failures, or stagnant execution... MineEvolve first uses Monitor to convert each subgoal execution into typed feedback, including state changes, inventory changes, failure types, progress signals, and stagnation indicators.",
  "horizon_span": "involve long prerequisite chains and are frequently disrupted by missing tools, blocked paths, GUI failures, or stagnant execution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286519800",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290782015",
  "title": "MirrorCraft: Paired Evaluation under Hidden Rule Changes in Minecraft",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 0,
  "publication_date": "2026-07-31",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "MirrorCraft",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "achieve one of three progression objectives per matched Vanilla/Mirror world pair; adapt to hidden server-side rule changes (recipes, drops, other mechanics) altered by a datapack; reach deterministic advancement milestones despite altered rules",
  "goal_origin": "mixed:given-up-front-objectives-with-hidden-environment-rule-changes",
  "decomposition": "set-of-independent",
  "interdependence": "Terrain, spawn, resource placement, objective, interface, and action budget remain matched within a Vanilla-Mirror pair except for the deliberately modified rule(s), so the Rule Intervention Effect isolates how a single changed rule affects goal achievement.",
  "n_goals": "five controlled biomes, six rule suites, three progression objectives, two model families, six agent configurations",
  "tracking_demand": "Agent must track task/advancement-milestone progress under its assigned rule suite while detecting and adapting to whichever server-side rule was silently modified in its Mirror world.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Among configurations evaluated without rule descriptions, ReAct achieves the highest pooled Mirror score; providing the exact rules yields only modest gains in average progress/completion across all three objectives (no human baseline reported).",
  "availability": null,
  "goal_span": "We evaluate task progress with deterministic advancement milestones and success rate and use the Rule Intervention Effect (RIE) to measure the performance change between matched Vanilla and Mirror worlds.",
  "horizon_span": "MirrorCraft includes five controlled biomes, six rule suites, three progression objectives, two model families, and six agent configurations under a shared Mineflayer interface.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290782015",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288257806",
  "title": "NARRA-Gym for Evaluating Interactive Narrative Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 0,
  "publication_date": "2026-05-08",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "NARRA-Gym",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "sustain a coherent, evolving story across multiple turns while adapting to a specific user persona; manage long-context state and pacing across the episode; maintain consistent character simulation and empathic personalization; optionally synthesize a story-grounded artifact at the end of the episode",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later turns' story construction, pacing interventions, and personalization decisions depend on the accumulated memory/state (story so far, persona model) built up over earlier turns in the same episode, so the agent must jointly manage all of these across the whole session rather than turn-by-turn in isolation.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain long-context story state, memory updates, pacing decisions, and an evolving user-persona model consistently across the whole interactive episode.",
  "scoring": "LLM-judge-rubric. The paper evaluates nine frontier LLMs using 'a controlled LLM-as-judge sweep over eight benchmark personas and a human evaluation,' explicit judge-based rating rather than a binary or milestone-based rubric; no subgoal-level partial credit distinct from these multi-dimension judge ratings is described.",
  "horizon_value": "each interactive episode lasts roughly 20 minutes (human evaluation)",
  "horizon_unit": "wall-clock-minutes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one complete interactive story episode)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce NARRA-Gym, an executable evaluation environment that turns a sparse emotional seed into a complete interactive story episode and logs the full model-in-the-loop trajectory, including story construction, memory updates, planning, pacing interventions, and optional artifact synthesis.",
  "horizon_span": "each interactive episode lasts roughly 20 minutes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288257806",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "wall-clock-minutes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289099032",
  "title": "OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 0,
  "publication_date": "2026-06-08",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "OmniGameArena",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "achieve a high score/objective in each of 12 distinct UE5 games (7 Solo, 3 PvP, 2 Coop); reflect on past-round performance to refine a bounded skill prompt across multiple rounds (Improvement Dynamics Curve); generalize a learned/refined skill to held-out task variants",
  "goal_origin": "mixed:given-up-front-game-objectives-plus-self-generated-skill-refinement",
  "decomposition": "sequential-chain",
  "interdependence": "Each reflection round's refined skill prompt is built from the trajectories generated by the previous round's episodes, so improvement at round r depends causally on what was learned and distilled in rounds 1..r-1, and final generalization is assessed on held-out task variants using the skill refined through this cross-round process.",
  "n_goals": "12 games (7 Solo, 3 PvP, 2 Coop); R=10 reflection rounds x K=5 episodes each = 50 episodes per agent-game evaluation",
  "tracking_demand": "The reflector must track trajectories and the persistent skill state across rounds, retaining what previously worked or failed, to progressively refine the bounded skill prompt rather than starting fresh each round.",
  "scoring": "continuous-reward. The paper reports 'cold-start leaderboard scores' plus the Improvement Dynamics Curve showing 'how the score evolves across reflection rounds, and how the learned skill behaves on held-out task variants' -- round-by-round score tracking rather than a single subgoal-checkpoint rubric.",
  "horizon_value": "R=10 reflection rounds, each comprising K=5 episodes (50 episodes total per agent-game IDC run)",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per agent-game IDC evaluation (10 rounds x 5 episodes each)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we address these gaps with OmniGameArena, a real-time benchmark of twelve newly built Unreal Engine 5 games spanning Solo (7), PvP (3), and Coop (2) with unified action interfaces, and the Improvement Dynamics Curve (IDC), an agentic-reflection harness in which a tool-using reflector LLM autonomously refines a bounded skill prompt across multiple rounds.",
  "horizon_span": "Each agent completes R=10 rounds of K=5 episodes under PDQ",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289099032",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287256539",
  "title": "Mastering PokeGym: Graph-Guided Multimodal Evolution at Test Time",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 1,
  "publication_date": "2026-04-09",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "PokeGym",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "complete a long-horizon task in a 3D open-world game using only visual observations (no game-state access); improve/adapt the agent's own configuration (perception, strategy, action set) across consecutive episodes of the same task (test-time learning); jointly optimize perception, reasoning, and control rather than a single modality in isolation",
  "goal_origin": "mixed:given-up-front-task-with-self-generated-test-time-improvement-goal",
  "decomposition": "hierarchical",
  "interdependence": "Because the same task is repeated across consecutive episodes for test-time learning, an agent's configuration choices (perception, strategy, action set) in earlier episodes constrain and inform what it tries in later episodes of the same task; the graph-guided framework explicitly tracks this cross-episode evolution.",
  "n_goals": "30 tasks derived from 10 quests; trajectories ranging from 30 to 220 environment steps",
  "tracking_demand": "The agent must track its own evolving configuration (which perception/strategy/action choices helped or hurt) across consecutive episodes of the same task, in addition to in-episode state, since the goal is test-time learning rather than a single fixed-policy run.",
  "scoring": "continuous-reward. G-EvoMAC achieves a 60.18% average success rate on PokeGym; the abstract does not describe subgoal-checkpoint partial credit distinct from this success-rate metric, though performance is tracked across consecutive episodes to measure improvement.",
  "horizon_value": "30 tasks derived from 10 quests, with trajectories ranging from 30 to 220 environment steps",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/trajectory (range across the 30 tasks); test-time learning further repeats each task across consecutive episodes",
  "horizon_stated": "yes",
  "headline_result": "G-EvoMAC achieves a 60.18% average success rate on PokeGym, outperforming the strongest baseline by over 11 percentage points; no human/expert baseline given in the abstract.",
  "availability": null,
  "goal_span": "we first introduce \\textbf{PokeGym}, a long-horizon benchmark built upon the 3D open-world game Pok\\'emon Legends: Z-A, where agents act from visual observations without access to game states, designed to evaluate an agent's ability to learn and adapt across consecutive episodes of the task.",
  "horizon_span": "PokeGym is a benchmark that contains 30 tasks derived from 10 quests, with trajectories ranging from 30 to 220 environment steps.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287256539",
  "provenance": "forward-citation,web-registry",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284597199",
  "title": "TowerMind: A Tower Defence Game Learning Environment and Benchmark for LLM as Agents",
  "year": 2026,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 5,
  "publication_date": "2026-01-09",
  "months_since_pub": 8,
  "citations_per_month": 0.62,
  "artifact_name": "TowerMind",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "perform macro-level strategic planning (tower placement, resource allocation) in a tower-defense scenario; perform micro-level tactical adaptation and action execution in response to incoming waves; succeed across each of five designed benchmark levels under different multimodal input settings; avoid hallucinating about game state while planning and acting",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Macro-level strategic decisions (where to place towers, how to allocate resources) constrain what micro-level tactical responses are available/effective against later waves, so a poor early strategic choice compounds into tactical difficulty in subsequent waves.",
  "n_goals": "five benchmark levels; evaluated under different multimodal input settings (pixel-based, textual, structured game-state)",
  "tracking_demand": "The agent must track macro-level game state (tower placements, resources, wave progression) and micro-level tactical details (unit/enemy positions) across the tower-defense match, while its outputs are additionally checked for hallucination relative to the true game state.",
  "scoring": "other:capability-and-hallucination-comparison-vs-human-experts-plus-classic-RL-baselines",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "There is a clear performance gap between LLMs and human experts across both capability and hallucination dimensions on TowerMind's five benchmark levels; LLMs show inadequate planning validation, lack of multifinality in decision-making, and inefficient action use. No single specific numeric score is given in the abstract.",
  "availability": null,
  "goal_span": "We design five benchmark levels to evaluate several widely used LLMs under different multimodal input settings. The results reveal a clear performance gap between LLMs and human experts across both capability and hallucination dimensions.",
  "horizon_span": "their inherent gameplay requires both macro-level strategic planning and micro-level tactical adaptation and action execution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284597199",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289098178",
  "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 1,
  "publication_date": "2026-06-06",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "alem",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "survive and grow within a long-horizon Craftax-like survival world (exploration, crafting, trading, combat); coordinate role allocation with teammates (soft specialisation); communicate to allocate roles and execute shared plans; solve procedurally generated coordination tasks of controllable difficulty",
  "goal_origin": "mixed:survival-coordination-framing-given-up-front-role-allocation-self-generated",
  "decomposition": "open-ended",
  "interdependence": "Team members' soft specialisation and communication choices affect which coordination tasks can be jointly solved; base survival tasks (exploration, crafting, trading, combat) share world resources across teammates, so one agent's resource use or role choice constrains others' options, and coordination difficulty is controllably scaled via procedurally generated tasks.",
  "n_goals": null,
  "tracking_demand": "Agents must track their own survival state (health, resources, crafted items), teammates' roles/communications, and progress on procedurally generated coordination tasks within a long-horizon survival world.",
  "scoring": "continuous-reward. Performance is measured via normalised return (current LLM agents average only ~6%) plus separate base-task reward vs. coordination reward, showing individual task competence does not imply coordination competence -- two distinct reward channels rather than one binary score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Current LLM agents average only ~6% normalised return on alem; zero-shot Gemini-3.1-Pro-High approaches MARL agents trained for one billion steps on the hardest coordination setting, while GPT-5.4-High achieves strong base-task reward but much lower coordination reward.",
  "availability": "https://github.com/alem-world/alem-env",
  "goal_span": "Alem embeds procedurally generated coordination tasks, soft specialisation, communication, and controllable coordination difficulty into a long-horizon survival world with exploration, crafting, trading, and combat.",
  "horizon_span": "they must coordinate with others over long horizons in open-ended interactive tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289098178",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "mixed",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276885285",
  "title": "AVA: Attentive VLM Agent for Mastering StarCraft II",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 3,
  "publication_date": "2025-03-07",
  "months_since_pub": 18,
  "citations_per_month": 0.17,
  "artifact_name": "AVACraft",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "complete micromanagement objectives (e.g. unit-level combat control); achieve coordination objectives among allied units/agents; execute strategic planning objectives across 21 StarCraft II scenarios",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Micromanagement decisions (unit-level control) feed into higher-level coordination among units, which in turn constrains what strategic plans are achievable; the paper explicitly frames these as three levels spanning micromanagement, coordination, and strategic planning.",
  "n_goals": "21 scenarios spanning micromanagement, coordination, and strategic planning",
  "tracking_demand": "The agent/policy must track unit-level state (health, position) for micromanagement, coordinate allocation across units for team objectives, and maintain a strategic plan across the scenario, using RGB visuals, natural-language observations, and structured state.",
  "scoring": "other:win-rate-percentage-marl-vs-zero-shot-vlm",
  "horizon_value": "300 seconds (5 minutes) max per episode, or earlier if victory conditions are met",
  "horizon_unit": "wall-clock-minutes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode",
  "horizon_stated": "yes",
  "headline_result": "MARL peaks at 19.3% win rate after 5M training steps, while VLMs achieve 75-90% zero-shot with human-aligned decisions, exposing training-efficiency vs. performance-ceiling trade-offs; no direct human baseline given.",
  "availability": "https://github.com/camel-ai/VLM-Play-StarCraft2",
  "goal_span": "AVACraft provides RGB visuals, natural language observations, and structured state information, enabling systematic comparison between training-based and zero-shot methods across 21 scenarios spanning micromanagement, coordination, and strategic planning.",
  "horizon_span": "Episodes terminate when 300 seconds pass or when victory conditions are met.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276885285",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "wall-clock-minutes",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "276647196",
  "title": "Collab-Overcooked: Benchmarking and Evaluating Large Language Models as Collaborative Agents",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 38,
  "publication_date": "2025-02-27",
  "months_since_pub": 19,
  "citations_per_month": 2.0,
  "artifact_name": "Collab-Overcooked",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "fulfill multiple simultaneous cooking orders/objectives via natural-language multi-agent coordination; actively collaborate and continuously adapt strategy as the shared kitchen task unfolds",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Cooking sub-tasks (chopping, cooking, plating, delivering) have precedence constraints and share kitchen resources/space, so agents must coordinate via natural-language communication to avoid conflicts and jointly complete orders within a time budget scaled to the task's optimal completion time.",
  "n_goals": "30 open-ended tasks across 6 complexity levels; time constraint set as optimal completion time scaled by a time-limit factor gamma (gamma=1.5 in experiments)",
  "tracking_demand": "Agents must track the shared kitchen's evolving state (ingredient/tool locations, in-progress dishes), coordinate via natural-language communication to avoid resource conflicts, and manage their actions within a per-task time budget derived from the optimal completion time.",
  "scoring": "other:process-oriented-metrics -- the benchmark introduces a spectrum of process-oriented evaluation metrics to assess the fine-grained collaboration capabilities of different LLM agents, fine-grained, process-level credit distinct from a single binary task-success score.",
  "horizon_value": "each task's time constraint is set as the optimal completion time scaled by a time-limit factor gamma (gamma=1.5 used in experiments); exact timestep counts vary by complexity level across the 30 tasks / 6 complexity levels",
  "horizon_unit": "agent-steps",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/YusaeMeow/Collab-Overcooked",
  "goal_span": "Collab-Overcooked extends existing benchmarks in two novel ways. First, it provides a multi-agent framework supporting diverse tasks and objectives and encourages collaboration through natural language communication. Second, it introduces a spectrum of process-oriented evaluation metrics to assess the fine-grained collaboration capabilities of different LLM agents, a dimension often overlooked in prior work.",
  "horizon_span": "Each task has a time constraint, set as the optimal completion time scaled by a time limit factor \u03b3.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276647196",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276902753",
  "title": "DSGBench: A Diverse Strategic Game Benchmark for Evaluating LLM-based Agents in Complex Decision-Making Environments",
  "year": 2025,
  "venue": "IEEE International Conference on Acoustics, Speech, and Signal Processing",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 21,
  "publication_date": "2025-03-08",
  "months_since_pub": 18,
  "citations_per_month": 1.17,
  "artifact_name": "DSGBench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "make long-term, multi-dimensional strategic decisions in each of six complex strategic games; adapt task difficulty and targets within customizable game settings",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Strategic games inherently chain decisions over time (long-term, multi-dimensional demands), and the paper's 'automated decision-tracking mechanism' explicitly analyzes 'the turning points in their strategies', implying later decisions are shaped by, and evaluated against, the trajectory of earlier ones.",
  "n_goals": "6 strategic games; 5 evaluation dimensions per game",
  "tracking_demand": "System tracks the agent's full decision trajectory across a game (via the automated decision-tracking mechanism) to identify behavior patterns and strategy turning points, and scores performance along five specific dimensions.",
  "scoring": "other:mixed \u2014 'a fine-grained evaluation scoring system which examines the decision-making capabilities by looking into the performance in five specific dimensions', i.e. multi-dimensional rubric-style scoring rather than a single binary win/loss.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Evaluation of six popular LLM agents reveals 'distinct strengths and limitations among various tasks' and 'systemic limitations in different LLMs' (no single numeric headline score or human baseline given).",
  "availability": null,
  "goal_span": "DSGBench employs a fine-grained evaluation scoring system which examines the decision-making capabilities by looking into the performance in five specific dimensions, offering a comprehensive assessment in a better-designed fashion.",
  "horizon_span": "it incorporates six complex strategic games which serve as ideal testbeds due to their long-term and multi-dimensional decision-making demands",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276902753",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276961240",
  "title": "Factorio Learning Environment",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 5,
  "publication_date": "2025-03-06",
  "months_since_pub": 18,
  "citations_per_month": 0.28,
  "artifact_name": "Factorio Learning Environment (FLE)",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "complete each of 8 fixed lab-play structured tasks; in open-play, build the largest possible factory on a procedurally generated map (an unbounded, self-scaling goal); scale automation from basic production to factories processing millions of resource units per second",
  "goal_origin": "mixed:given-up-front-lab-play-and-open-ended-open-play",
  "decomposition": "other:hierarchical-in-lab-play,-open-ended-in-open-play",
  "interdependence": "Building larger factories requires production chains where later automation goals (e.g., electronic-circuit manufacturing) depend on earlier infrastructure (e.g., electric-powered drilling) already being in place.",
  "n_goals": "8 fixed tasks in lab-play; open-play is unbounded (no fixed count)",
  "tracking_demand": "Agent must track resource/production-chain state, spatial factory layout, and automation progress as goals scale from basic automation to factories processing millions of resource units per second.",
  "scoring": "other:task-success-in-lab-play-plus-qualitative-automation-progress-in-open-play",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We provide two settings: (1) lab-play consisting of eight structured tasks with fixed resources, and (2) open-play with the unbounded task of building the largest factory on an procedurally generated map.",
  "horizon_span": "FLE provides exponentially scaling challenges -- from basic automation to complex factories processing millions of resource units per second.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276961240",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281080032",
  "title": "FlashAdventure: A Benchmark for GUI Agents Solving Full Story Arcs in Diverse Adventure Games",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 2,
  "publication_date": "2025-09-01",
  "months_since_pub": 12,
  "citations_per_month": 0.17,
  "artifact_name": "FlashAdventure",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "complete each of a game's predefined success milestones in the correct narrative order; remember and act on earlier gameplay information to bridge the observation-behavior gap; complete the full story arc of one of 34 diverse Flash-based adventure games",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "CUA-as-a-Judge holds predefined success milestones per game and verifies them in sequence as the story progresses, so completing a later milestone typically requires having correctly retained and acted on clue information established by an earlier milestone (the 'observation-behavior gap').",
  "n_goals": "34 Flash-based adventure games spanning mystery/detective, hidden object, room escape, visual novel, and simulation subgenres",
  "tracking_demand": "The agent must remember earlier gameplay clues/information (long-term clue memory) and correctly act on them at later points in the story to progress through predefined milestones toward full story-arc completion.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "GPT-4o and UI-TARS achieve 0% success completing full story arcs; COAST (built on Claude 3.7 Sonnet) reaches a 5.88% success rate and 19.89% milestone completion rate; no human baseline given.",
  "availability": null,
  "goal_span": "we introduce FlashAdventure, a benchmark of 34 Flash-based adventure games designed to test full story arc completion and tackle the observation-behavior gap: the challenge of remembering and acting on earlier gameplay information. We also propose CUA-as-a-Judge, an automated gameplay evaluator, and COAST, an agentic framework leveraging long-term clue memory to better plan and solve sequential tasks.",
  "horizon_span": "COAST ... reaching a 5.88% success rate and a 19.89% milestone completion rate",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281080032",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280676531",
  "title": "HeroBench: A Benchmark for Long-Horizon Planning and Structured Reasoning in Virtual Worlds",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 6,
  "publication_date": "2025-08-18",
  "months_since_pub": 13,
  "citations_per_month": 0.46,
  "artifact_name": "HeroBench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "select numerically feasible equipment given resource/stat constraints; reason over multi-level crafting and resource dependencies; execute hundreds to thousands of actions as a single coherent end-to-end plan; succeed in numeric combat simulation against scalable, adversarially-distracted difficulty",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Crafting is multi-level, so obtaining a higher-tier item requires resources/sub-items produced by lower-tier crafting steps, and the resulting plan must remain numerically feasible (equipment/combat stats) end-to-end; the benchmark integrates symbolic planning, numeric simulation, and spatial reasoning into one execution.",
  "n_goals": null,
  "tracking_demand": "The agent must track multi-level crafting/resource dependencies, numeric feasibility of equipment choices, and spatial state across a single end-to-end plan comprising hundreds to thousands of actions, verified by simulation-based success and fine-grained progress metrics.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": "hundreds to thousands",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/plan",
  "horizon_stated": "yes",
  "headline_result": "An evaluation of 25 state-of-the-art LLMs reveals large performance disparities; reasoning models perform substantially better but no model reliably solves the hardest tasks; no explicit human baseline given.",
  "availability": null,
  "goal_span": "Tasks require models to select numerically feasible equipment, reason over multi-level crafting and resource dependencies, and execute hundreds to thousands of actions as a single end-to-end plan. HeroBench integrates symbolic planning, numeric combat simulation, spatial reasoning, and resource management ... HeroBench evaluates executable plans through simulation, enabling both success-based and fine-grained progress metrics, as well as detailed failure mode analysis.",
  "horizon_span": "execute hundreds to thousands of actions as a single end-to-end plan",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280676531",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282102736",
  "title": "EvoTest: Evolutionary Test-Time Learning for Self-Improving Agentic Systems",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 31,
  "publication_date": "2025-10-15",
  "months_since_pub": 11,
  "citations_per_month": 2.82,
  "artifact_name": "J-TTL (Jericho Test-Time Learning)",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "solve an interactive-fiction game requiring many in-game puzzle subgoals (e.g., in Detective and Library) within an episode; improve performance from one episode to the next by adapting across consecutive playthroughs of the same game",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Progress within an episode depends on solving game puzzles in the correct order, and across episodes, later playthroughs are conditioned on what was learned (memory, prompt revisions, tuned hyperparameters) from earlier episodes of the same game.",
  "n_goals": null,
  "tracking_demand": "The Actor Agent (and the Evolver Agent that analyzes transcripts) must track in-episode state-action choices and effective strategies, then carry a revised configuration (prompt, memory, hyperparameters, tool-use routines) forward to the next episode of the same game.",
  "scoring": "continuous-reward - J-TTL measures whether an agent's performance improves from one episode to the next on the same game; abstract highlights winning entire games (Detective, Library) as a milestone-like outcome but does not describe a formal in-episode subgoal-checkpoint credit scheme beyond overall game score/win.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "EvoTest is the only method capable of winning two games (Detective and Library) on J-TTL, while all baselines (reflection, memory-only, and more complex online fine-tuning methods) fail to win any; no human/expert baseline is given.",
  "availability": null,
  "goal_span": "J-TTL is a new evaluation setup where an agent must play the same game for several consecutive episodes, attempting to improve its performance from one episode to the next.",
  "horizon_span": "J-TTL is a new evaluation setup where an agent must play the same game for several consecutive episodes, attempting to improve its performance from one episode to the next.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282102736",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278129561",
  "title": "Collaborating Action by Action: A Multi-agent LLM Framework for Embodied Reasoning",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 25,
  "publication_date": "2025-04-24",
  "months_since_pub": 17,
  "citations_per_month": 1.47,
  "artifact_name": "MineCollab",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "control characters collaboratively in open-world Minecraft to complete complex embodied reasoning tasks; delegate sub-tasks between collaborating agents via natural-language communication; share and update task-completion plans among agents as the task progresses",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Completing the overall Minecraft task requires delegated sub-tasks to be correctly assigned and communicated between agents; performance drops when agents must communicate detailed task-completion plans, showing that later collaborative steps are contingent on earlier, correctly shared plan information.",
  "n_goals": null,
  "tracking_demand": "Agents must track their own sub-task assignment, what has been communicated to/from collaborators, and the shared task-completion plan's current state across the collaborative embodied task.",
  "scoring": "other:dimension-specific-collaboration-metrics - tests different dimensions of embodied and collaborative reasoning, finding communication (not embodied action itself) is the primary bottleneck, with performance dropping up to 15% when detailed plans must be communicated; a multi-dimensional evaluation rather than a single pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Agent performance drops as much as 15% when agents are required to communicate detailed task completion plans, identifying efficient natural language communication as the primary bottleneck to effective collaboration; no explicit human/expert baseline is given.",
  "availability": "https://mindcraft-minecollab.github.io/",
  "goal_span": "we introduce MINDcraft, an easily extensible platform built to enable LLM agents to control characters in the open-world game of Minecraft; and MineCollab, a benchmark to test the different dimensions of embodied and collaborative reasoning.",
  "horizon_span": "This work studies how LLMs can adaptively collaborate to perform complex embodied reasoning tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278129561",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276250438",
  "title": "LLM-Powered Decentralized Generative Agents with Adaptive Hierarchical Knowledge Graph for Cooperative Planning",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 17,
  "publication_date": "2025-02-08",
  "months_since_pub": 19,
  "citations_per_month": 0.89,
  "artifact_name": "Multi-agent Crafter environment (DAMCS)",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "achieve open-world survival/crafting objectives cooperatively with other agents; share and act on relevant information from past interactions via a hierarchical knowledge-graph memory; coordinate via structured communication to avoid redundant or conflicting actions among 2-6 agents",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Crafting/survival objectives in the Crafter-style world have prerequisite chains, and cooperating agents additionally share a limited action/resource budget and must coordinate to avoid redundant work, which is why more agents achieving the same goal need proportionally fewer total steps when memory/communication are used well.",
  "n_goals": null,
  "tracking_demand": "Each agent must track its own past experience via a hierarchical knowledge-graph memory and selectively communicate relevant facts to teammates rather than sharing full history, in order to reach a shared long-term goal efficiently.",
  "scoring": "other:not-stated \u2014 the abstract reports task efficiency via step counts and 'collaboration' quality but does not describe explicit subgoal-checkpoint credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Two-agent DAMCS reaches the same goal with 63% fewer steps than single-agent, and six-agent with 74% fewer steps, versus MARL/LLM baselines (no single 'best model' figure or human comparison stated).",
  "availability": "https://happyeureka.github.io/damcs",
  "goal_span": "Compared to single-agent scenarios, the two-agent scenario achieves the same goal with 63% fewer steps, and the six-agent scenario with 74% fewer steps, highlighting the importance of adaptive memory and structured communication in achieving long-term goals.",
  "horizon_span": "the two-agent scenario achieves the same goal with 63% fewer steps, and the six-agent scenario with 74% fewer steps",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276250438",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "282057761",
  "title": "ParaCook: On Time-Efficient Planning for Multi-Agent Systems",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 1,
  "publication_date": "2025-10-13",
  "months_since_pub": 11,
  "citations_per_month": 0.09,
  "artifact_name": "ParaCook",
  "artifact_kind": "environment/simulator",
  "domain": "open-world-game",
  "goal_types": "prepare and deliver each of several dish orders correctly; coordinate parallel/asynchronous sub-tasks (e.g. chopping, cooking, plating) across multiple agents to minimize completion time; avoid collisions/conflicts over shared kitchen resources while parallelizing actions",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Sub-tasks for a dish have precedence constraints (e.g. an ingredient must be chopped before cooking, cooked before plating) and share physical kitchen resources (stations, tools), so agents must coordinate to parallelize compatible steps without conflicting over the same resource.",
  "n_goals": "adjustable-complexity scalable evaluation framework; number of simultaneous dish/order goals not fixed in the abstract",
  "tracking_demand": "Agents must track the preparation-stage state of each in-progress dish (which precedence steps are done), coordinate with other agents to avoid resource conflicts, and jointly minimize overall completion time across simultaneous orders.",
  "scoring": "continuous-reward -- performance is measured via time-efficiency of the overall plan (completion time), not through a subgoal-checkpoint partial-credit rubric described in the abstract; no explicit statement of intra-task partial credit is given.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/zsq259/ParaCook",
  "goal_span": "Inspired by the Overcooked game, ParaCook provides an environment for various challenging interaction planning of multi-agent systems that are instantiated as cooking tasks, with a simplified action space to isolate the core challenge of strategic parallel planning.",
  "horizon_span": "ParaCook provides a scalable evaluation framework with adjustable complexity, establishing a foundation for developing and assessing time efficiency-aware multi-agent planning.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282057761",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280649965",
  "title": "SC2Arena and StarEvolve: Benchmark and Self-Improvement Framework for LLMs in Complex Decision-Making Tasks",
  "year": 2025,
  "venue": null,
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 2,
  "publication_date": "2025-08-14",
  "months_since_pub": 13,
  "citations_per_month": 0.15,
  "artifact_name": "SC2Arena / StarEvolve",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "manage a full StarCraft II game across the complete game context (all playable races); operate over diverse, low-level action spaces rather than a simplified/reduced action space; solve spatial reasoning challenges via text-based observations; integrate strategic planning with tactical execution (Planner-Executor-Verifier structure)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "StarEvolve's Planner-Executor-Verifier structure explicitly breaks gameplay into a hierarchy where tactical execution must serve the strategic plan and is checked/corrected by a verifier, so a failure at the tactical (execution) level can require iterative self-correction that feeds back into future planning and fine-tuning on high-quality gameplay data.",
  "n_goals": null,
  "tracking_demand": "The agent must track the complete game state (all playable races, diverse action spaces) and its own strategic plan versus tactical execution outcomes, using a scoring system to select high-quality training samples for continuous improvement.",
  "scoring": "other:strategic-planning-performance-analysis. The abstract reports that 'StarEvolve achieves superior performance in strategic planning' via 'comprehensive analysis using SC2Arena,' without describing a formal subgoal-checkpoint or milestone rubric distinct from in-game outcome/planning-quality measures.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we present SC2Arena, a benchmark that fully supports all playable races, low-level action spaces, and optimizes text-based observations to tackle spatial reasoning challenges. Complementing this, we introduce StarEvolve, a hierarchical framework that integrates strategic planning with tactical execution, featuring iterative self-correction and continuous improvement via fine-tuning on high-quality gameplay data.",
  "horizon_span": "SC2Arena, a benchmark that fully supports all playable races, low-level action spaces, and optimizes text-based observations to tackle spatial reasoning challenges",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280649965",
  "provenance": "web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280280196",
  "title": "StarDojo: Benchmarking Open-Ended Behaviors of Agentic Multimodal LLMs in Production-Living Simulations with Stardew Valley",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 5,
  "publication_date": "2025-07-10",
  "months_since_pub": 14,
  "citations_per_month": 0.36,
  "artifact_name": "StarDojo",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "perform livelihood production activities (farming, crafting); engage in social interactions to build relationships within the community; complete tasks across five key domains: farming, crafting, exploration, combat, and social interactions",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Production activities (farming/crafting) and social-relationship building must be pursued simultaneously and can compete for the same limited in-game time/attention, since 'agents are tasked to perform essential livelihood activities... while simultaneously engaging in social interactions'.",
  "n_goals": "1,000 curated tasks (100-task representative subset) across 5 domains",
  "tracking_demand": "Agent must track progress across both production (farming/crafting/exploration/combat) and social-relationship goals concurrently, within a unified interface supporting parallel environment instances.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports an overall task success rate (12.7% for GPT-4.1) but does not describe explicit subgoal-checkpoint partial credit across the five domains.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best-performing model GPT-4.1 achieves only a 12.7% success rate, primarily due to challenges in visual understanding, multimodal reasoning and low-level manipulation (no human baseline given).",
  "availability": null,
  "goal_span": "In StarDojo, agents are tasked to perform essential livelihood activities such as farming and crafting, while simultaneously engaging in social interactions to establish relationships within a vibrant community.",
  "horizon_span": "StarDojo features 1,000 meticulously curated tasks across five key domains: farming, crafting, exploration, combat, and social interactions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280280196",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "279402292",
  "title": "StoryBench: A Dynamic Benchmark for Evaluating Long-Term Memory with Multi Turns",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 13,
  "publication_date": "2025-06-16",
  "months_since_pub": 15,
  "citations_per_month": 0.87,
  "artifact_name": "StoryBench (2025, interactive-fiction based)",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "correctly retain and recall facts/state established earlier in a branching narrative (knowledge retention); reason over sequences of narrative events to infer state changes and causal dependencies (sequential reasoning); under one setting, trace back and revise earlier choices after a failure is detected",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "Each choice triggers cascading dependencies across multi-turn interactions, so an agent's later reasoning/recall must remain consistent with the specific branch of prior choices it actually took, and one evaluation setting explicitly requires tracing back to revise earlier choices after a failure.",
  "n_goals": "narrative dataset with 311 scene nodes and 86 choice nodes (spanning the game's prologue through Chapter 5)",
  "tracking_demand": "The agent must retain established narrative facts and track which branch of the hierarchical decision tree it is on, recognizing cascading state dependencies across many turns, including (in one setting) needing to trace back and revise an earlier choice after failure.",
  "scoring": "other:two-setting-evaluation -- one setting gives immediate feedback upon an incorrect decision while the other requires independently tracing back and revising earlier choices after failure, so credit differs by setting rather than being a single uniform binary score; the abstract does not state a combined numeric rubric.",
  "horizon_value": "narrative dataset spans 311 scene nodes and 86 choice nodes, extending from the game's prologue through Chapter 5",
  "horizon_unit": "other:narrative-nodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "dataset-wide scale (whole narrative graph), not a single fixed per-episode turn count",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we propose a novel benchmark framework based on interactive fiction games, featuring dynamically branching storylines with complex reasoning structures. These structures simulate real-world scenarios by requiring LLMs to navigate hierarchical decision trees, where each choice triggers cascading dependencies across multi-turn interactions. Our benchmark emphasizes two distinct settings to test reasoning complexity: one with immediate feedback upon incorrect decisions, and the other requiring models to independently trace back and revise earlier choices after failure.",
  "horizon_span": "We construct a narrative dataset based on the interactive fiction game The Invisible Guardian, encompassing 311 scene nodes and 86 choice nodes as captured in our structured JSON format.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279402292",
  "provenance": "asta-find,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277954804",
  "title": "TALES: Text Adventure Learning Environment Suite",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 11,
  "publication_date": "2025-04-19",
  "months_since_pub": 17,
  "citations_per_month": 0.65,
  "artifact_name": "TALES",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "complete puzzle/quest objectives within a synthetic or human-written text-adventure game via sequential decision-making; maintain structured reasoning over the accumulated context history to determine the next best action",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each move changes the game's world state (inventory, location, cooked ingredients), so later moves' validity and score contribution depend on the cumulative sequence of earlier moves; harder difficulty levels require substantially longer valid move sequences (44 vs. 7 moves) to reach the max score.",
  "n_goals": "diverse collection of synthetic and human-written text-adventure games; CookingWorld difficulty levels span from 7 moves (max score 3, level 1) to 44 moves (max score 11, level 10)",
  "tracking_demand": "The agent must track accumulated world state (inventory, location, prior actions) across a sequential-decision game, determining the next best action via structured reasoning over the context history, with harder levels requiring correctly sustaining this over up to 44 moves.",
  "scoring": "continuous-reward -- each game level has a 'max score' accumulated via correct moves (e.g. max score 11 at level 10), a continuous, level-scaled reward rather than one binary pass/fail; even top LLM-driven agents fail to achieve 15% on games designed for human enjoyment.",
  "horizon_value": "CookingWorld difficulty level 1 can be solved in 7 moves (max score 3), while level 10 requires 44 moves (max score 11)",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per game level)",
  "horizon_stated": "yes",
  "headline_result": "Even the top LLM-driven agents fail to achieve 15% on games designed for human enjoyment, despite an impressive showing on synthetic games; no specific single best-model number vs. a human baseline is given beyond this qualitative framing.",
  "availability": "https://microsoft.github.io/tale-suite",
  "goal_span": "We introduce TALES, a diverse collection of synthetic and human-written text-adventure games designed to challenge and evaluate diverse reasoning capabilities. We present results over a range of LLMs, open- and closed-weights, performing a qualitative analysis on the top performing models. Despite an impressive showing on synthetic games, even the top LLM-driven agents fail to achieve 15% on games designed for human enjoyment.",
  "horizon_span": "Difficulty level 1 can be solved in 7 moves with a max score of 3, while level 10 requires 44 moves with a max score of 11.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277954804",
  "provenance": "asta-find,parametric",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "actions",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "280401732",
  "title": "TextQuests: How Good are LLMs at Text-Based Video Games?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 11,
  "publication_date": "2025-07-31",
  "months_since_pub": 14,
  "citations_per_month": 0.79,
  "artifact_name": "TextQuests",
  "artifact_kind": "benchmark",
  "domain": "text-game-IF",
  "goal_types": "solve multi-puzzle interactive-fiction adventures with inventory/location/puzzle dependencies via trial-and-error; operate autonomously using only intrinsic long-context reasoning with no external tools; sustain self-directed reasoning across a long, growing context within a single interactive session",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Puzzles in the Infocom-suite adventures typically require earlier sub-goals (e.g., obtaining an item) to be solved before later ones become solvable, all within the growing context of a single session, with no external tool access permitted.",
  "n_goals": null,
  "tracking_demand": "Agent must track inventory, location, and puzzle-dependency state purely via long-context reasoning across up to hundreds of precise actions within a single, continuous interactive session, without external tool assistance.",
  "scoring": "other:not-specified-in-abstract",
  "horizon_value": "hundreds of actions (human playtime over 30 hours)",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (per full game playthrough/session)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://textquests.ai",
  "goal_span": "we introduce TextQuests, a benchmark based on the Infocom suite of interactive fiction games. These text-based adventures, which can take human players over 30 hours and require hundreds of precise actions to solve, serve as an effective proxy for evaluating AI agents on focused, stateful tasks.",
  "horizon_span": "These text-based adventures, which can take human players over 30 hours and require hundreds of precise actions to solve",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280401732",
  "provenance": "asta-find,parametric",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "278886118",
  "title": "VideoGameBench: Can Vision-Language Models complete popular video games?",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 23,
  "publication_date": "2025-05-23",
  "months_since_pub": 16,
  "citations_per_month": 1.44,
  "artifact_name": "VideoGameBench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "complete each of 10 popular 1990s video games end-to-end from raw visual input and high-level objective/control descriptions; generalize to 3 secret/unseen games not disclosed in advance; operate under real-time inference-latency constraints (or in a paused Lite setting)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "None is described beyond within-game progression; each of the 10 games is completed independently end-to-end, with only the shared real-time inference-latency constraint (removed in the Lite variant) linking how much progress a model can make.",
  "n_goals": "10 popular video games (3 kept secret)",
  "tracking_demand": "Agent must track in-game state (perception, spatial navigation, memory) purely from raw visual input across an entire game playthrough, without game-specific scaffolding or auxiliary information.",
  "scoring": "other:percent-of-game-completed",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best-performing models Gemini 2.5 Pro and Claude 3.7 Sonnet complete only 0.48% of VideoGameBench and 1.6% of VideoGameBench Lite; no human baseline given.",
  "availability": null,
  "goal_span": "we introduce VideoGameBench, a benchmark consisting of 10 popular video games from the 1990s that VLMs directly interact with in real-time. VideoGameBench challenges models to complete entire games with access to only raw visual inputs and a high-level description of objectives and controls, a significant departure from existing setups that rely on game-specific scaffolding and auxiliary information. We keep three of the games secret to encourage solutions that generalize to unseen environments.",
  "horizon_span": "The best performing models, Gemini 2.5 Pro and Claude 3.7 Sonnet, complete only 0.48% of VideoGameBench and 1.6% of VideoGameBench Lite.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278886118",
  "provenance": "parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279318802",
  "title": "WGSR-Bench: Wargame-based Game-theoretic Strategic Reasoning Benchmark for Large Language Models",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 5,
  "publication_date": "2025-06-12",
  "months_since_pub": 15,
  "citations_per_month": 0.33,
  "artifact_name": "WGSR-Bench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "achieve environmental situation awareness of a dynamic wargame scenario; model opponent risk accurately; generate a policy/action plan integrating awareness and risk modeling (the S-POE architecture)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Policy generation depends on accurate environmental situation awareness and opponent risk modeling completed earlier in the same wargame scenario; errors in awareness or opponent modeling propagate into poor policy choices.",
  "n_goals": "three core tasks (environmental situation awareness, opponent risk modeling, policy generation) forming the S-POE architecture",
  "tracking_demand": "Agent must track evolving battlefield/environmental state, model an adversary's likely behavior, and integrate both into policy generation within a single wargame scenario characterized by environmental uncertainty and adversarial dynamics.",
  "scoring": "other:comprehensive-strategy-reasoning-assessment",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "WGSR-Bench designs test samples around three core tasks, i.e., Environmental situation awareness, Opponent risk modeling and Policy generation, which serve as the core S-POE architecture, to systematically assess main abilities of strategic reasoning. Finally, an LLM-based wargame agent is designed to integrate these parts for a comprehensive strategy reasoning assessment.",
  "horizon_span": "Wargame, a quintessential high-complexity strategic scenario, integrates environmental uncertainty, adversarial dynamics, and non-unique strategic choices, making it an effective testbed for assessing LLMs' capabilities in multi-agent decision-making, intent inference, and counterfactual reasoning.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279318802",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279402668",
  "title": "WereWolf-Plus: An Update of Werewolf Game setting Based on DSGBench",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "text-game-IF",
  "citation_count": 0,
  "publication_date": "2025-06-15",
  "months_since_pub": 15,
  "citations_per_month": 0.0,
  "artifact_name": "WereWolf-Plus",
  "artifact_kind": "arena/leaderboard",
  "domain": "text-game-IF",
  "goal_types": "role-specific deduction/elimination objectives (e.g. werewolves eliminate villagers, Seer identifies werewolves, Witch/Hunter/Guard/Sheriff use special role abilities) tracked across the game; sustain social influence/cooperation or deception consistently across repeated day/night rounds",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Role-specific objectives interact through shared game state (votes, eliminations, revealed information), so an earlier round's outcome directly constrains which roles/objectives remain viable in later rounds, and deception/trust built in early rounds affects later persuasion success.",
  "n_goals": "customizable roles (Seer, Witch, Hunter, Guard, Sheriff, werewolves, villagers); comprehensive quantitative evaluation metrics for all special roles, werewolves, and the sheriff, across multiple models/methods",
  "tracking_demand": "Each agent must track the current game phase (day/night), revealed information, its own and others' inferred roles, and prior votes/eliminations across the game's rounds, adapting its strategy (cooperation, deception, or deduction) accordingly.",
  "scoring": "other:comprehensive-role-based-metrics -- a comprehensive set of quantitative evaluation metrics for all special roles, werewolves, and the sheriff, plus enriched dimensions for reasoning ability, cooperation capacity, and social influence, i.e. multi-role, multi-dimensional credit rather than a single binary win/loss score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/MinstrelsyXia/WereWolfPlus",
  "goal_span": "we propose WereWolf-Plus, a multi-model, multi-dimensional, and multi-method benchmarking platform for evaluating multi-agent strategic reasoning in the Werewolf game. The platform offers strong extensibility, supporting customizable configurations for roles such as Seer, Witch, Hunter, Guard, and Sheriff, along with flexible model assignment and reasoning enhancement strategies for different roles. In addition, we introduce a comprehensive set of quantitative evaluation metrics for all special roles, werewolves, and the sheriff, and enrich the assessment dimensions for agent reasoning ability, cooperation capacity, and social influence.",
  "horizon_span": "we propose WereWolf-Plus, a multi-model, multi-dimensional, and multi-method benchmarking platform for evaluating multi-agent strategic reasoning in the Werewolf game.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279402668",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278782258",
  "title": "lmgame-Bench: How Good are LLMs at Playing Games?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "open-world-game",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "open-world-game",
  "citation_count": 40,
  "publication_date": "2025-05-21",
  "months_since_pub": 16,
  "citations_per_month": 2.5,
  "artifact_name": "lmgame-Bench",
  "artifact_kind": "benchmark",
  "domain": "open-world-game",
  "goal_types": "complete platformer-game objectives requiring perception and timing; solve puzzle-game objectives requiring planning; progress narrative-game objectives requiring memory of prior game state; operate reliably despite brittle vision perception, prompt sensitivity, and potential data contamination",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each game in the suite probes a distinct blend of perception, memory, and planning capabilities, so an agent's performance on one game does not transfer straightforwardly to another; the paper notes reinforcement learning on a single game transfers to unseen games and external planning tasks, implying some shared underlying skill despite the games' surface independence.",
  "n_goals": null,
  "tracking_demand": "The agent must track in-game state (via lightweight perception and memory scaffolds) appropriate to each game's genre -- platformer timing/position, puzzle constraint state, or narrative progress -- delivered through a unified Gym-style API.",
  "scoring": "other:per-game-score-via-unified-gym-style-api",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Performing RL on a single game from lmgame-Bench transfers both to unseen games and to external planning tasks; no single numeric headline figure given.",
  "availability": "https://github.com/lmgame-org/GamingAgent/lmgame-bench",
  "goal_span": "LMGame-Bench features a suite of platformer, puzzle, and narrative games delivered through a unified Gym-style API and paired with lightweight perception and memory scaffolds, and is designed to stabilize prompt variance and remove contamination. Across 13 leading models, we show lmgame-Bench is challenging while still separating models well. Correlation analysis shows that every game probes a unique blend of capabilities often tested in isolation elsewhere.",
  "horizon_span": "a suite of platformer, puzzle, and narrative games delivered through a unified Gym-style API",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278782258",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286224440",
  "title": "AMemGym: Interactive Memory Benchmarking for Assistants in Long-Horizon Conversations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 16,
  "publication_date": "2026-03-02",
  "months_since_pub": 6,
  "citations_per_month": 2.67,
  "artifact_name": "AMemGym",
  "artifact_kind": "environment/simulator",
  "domain": "personal-assistant-memory",
  "goal_types": "answer state-dependent questions correctly using accumulated conversational context; track an evolving simulated-user state across a long-horizon conversation; adapt personalization/memory strategies as latent user state evolves through role-play",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "State-dependent questions depend on earlier state-evolution steps within the same trajectory; correctly answering later questions requires having tracked state changes revealed only through free-form interaction.",
  "n_goals": null,
  "tracking_demand": "Agent must maintain a consistent model of the simulated user's evolving latent state (from a predefined user profile plus a state-evolution trajectory), exposed only through free-form dialogue, to answer later state-dependent questions.",
  "scoring": "other:structured-metrics-not-fully-specified-in-abstract",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "AMemGym employs structured data sampling to predefine user profiles, state-dependent questions, and state evolution trajectories, enabling cost-effective generation of high-quality, evaluation-aligned interactions. LLM-simulated users expose latent states through role-play while maintaining structured state consistency.",
  "horizon_span": "Long-horizon interactions between users and LLM-based assistants necessitate effective memory management, yet current approaches face challenges in training and evaluation of memory.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286224440",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286224603",
  "title": "ASTRA-bench: Evaluating Tool-Use Agent Reasoning and Action Planning with Personal User Context",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 10,
  "publication_date": "2026-03-02",
  "months_since_pub": 6,
  "citations_per_month": 1.67,
  "artifact_name": "ASTRA-bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "ground reasoning in time-evolving personal context (longitudinal life events) to resolve a user's current intent; orchestrate reliable multi-step tool-use plans conditioned on that evolving personal context; correctly handle user intents annotated by referential, functional, and informational complexity",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each scenario's personal context evolves over time (longitudinal life events), so correctly resolving referential/functional/informational complexity in a given intent depends on correctly tracking earlier context updates for that protagonist.",
  "n_goals": "2,413 scenarios across four protagonists",
  "tracking_demand": "Agent must track a protagonist's time-evolving personal context (longitudinal life events) and ground tool arguments/reasoning in that evolving context across a scenario.",
  "scoring": "other:complexity-stratified-accuracy - performance is broken down by referential/functional/informational complexity tiers; abstract identifies argument generation as the primary bottleneck rather than describing an explicit subgoal-checkpoint credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "State-of-the-art models (e.g., Claude-4.5-Opus, DeepSeek-V3.2) show significant performance degradation under high-complexity conditions, with argument generation as the primary bottleneck; no specific numeric top score or human baseline is given.",
  "availability": null,
  "goal_span": "We present ASTRA-bench (Assistant Skills in Tool-use, Reasoning \\&Action-planning), a benchmark that uniquely unifies time-evolving personal context with an interactive toolbox and complex user intents.",
  "horizon_span": "Our event-driven pipeline generates 2,413 scenarios across four protagonists, grounded in longitudinal life events and annotated by referential, functional, and informational complexity.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286224603",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285101952",
  "title": "AgentIF-OneDay: A Task-level Instruction-Following Benchmark for General AI Agents in Daily Scenarios",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 7,
  "publication_date": "2026-01-28",
  "months_since_pub": 8,
  "citations_per_month": 0.88,
  "artifact_name": "AgentIF-OneDay",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "adhere to an explicit, complex, user-given workflow (Open Workflow Execution); infer implicit/latent instructions from attached files (Latent Instruction); modify or expand upon work already produced earlier in the task (Iterative Refinement); deliver a correct, tangible file-based result, not just a dialogue answer",
  "goal_origin": "mixed:given-up-front-plus-implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Iterative Refinement tasks require modifying or expanding upon ongoing work, so later steps depend on correctly understanding and preserving the state of earlier-produced file-based results; Latent Instruction tasks require inferring implicit requirements from attachments before later explicit steps can be correctly executed.",
  "n_goals": "104 tasks covering 767 scoring points, across 3 user-centric categories (Open Workflow Execution, Latent Instruction, Iterative Refinement)",
  "tracking_demand": "The agent must track the explicit and inferred requirements of a task across multiple attachments and any prior output it has already produced, so that later refinement or workflow steps remain consistent with earlier ones.",
  "scoring": "milestone-rubric. The benchmark uses 'instance-level rubrics' totaling 767 scoring points across 104 tasks, genuine per-criterion (subgoal-level) partial credit rather than a single binary pass/fail, evaluated via an LLM-based verification pipeline aligned with human judgment (80.1% agreement).",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "AgentIF-OneDay comprises 104 tasks covering 767 scoring points... We employ instance-level rubrics and a refined evaluation pipeline that aligns LLM-based verification with human judgment, achieving an 80.1% agreement rate using Gemini-3-Pro.",
  "horizon_span": "AgentIF-OneDay comprises 104 tasks covering 767 scoring points.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285101952",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290844740",
  "title": "OneDayAgent: Towards a Long-Horizon Harness for Autonomous Agents",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-08-04",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "AgentIF-OneDay (evaluated via the OneDayAgent harness)",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "decompose an open-ended everyday request spanning work, study, and life into bounded subtasks; preserve goals and constraints across many steps while navigating heterogeneous tools and attachments, avoiding goal drift and state loss; verify and repair the final deliverable against context-overflow and other failure modes",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "hierarchical",
  "interdependence": "The open-ended request is decomposed into bounded subtasks whose execution must preserve goals/constraints established at decomposition time; execution memory maintained under context pressure earlier in the process constrains what verification/repair is possible for the final deliverable.",
  "n_goals": "104 tasks in the AgentIF-OneDay evaluation set",
  "tracking_demand": "Harness must maintain execution memory of goals/constraints and subtask progress under context pressure across an open-ended, long-horizon, cross-environment, multimodal request, and verify/repair the final deliverable at the end.",
  "scoring": "continuous-reward - reports an overall score (0.821 with the GLM-5.2 backend) on AgentIF-OneDay; abstract does not explicitly describe subgoal-level checkpoint partial credit, though the harness's own internal decomposition into 'bounded subtasks' is not reflected in the reported top-line score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "With the GLM-5.2 backend, OneDayAgent sets a new state of the art on AgentIF-OneDay with an overall score of 0.821; the same harness generalizes across five backend LLMs from three model families without tuning; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "OneDayAgent turns an open-ended request into a managed execution process that decomposes tasks into bounded subtasks, maintains execution memory under context pressure, and verifies and repairs the final deliverable.",
  "horizon_span": "These tasks are long-horizon, cross-environment, and multimodal, forcing the agent to preserve goals and constraints across many steps while navigating heterogeneous tools and attachments.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290844740",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288671771",
  "title": "Your Agents Are Aging Too: Agent Lifespan Engineering for Deployed Systems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 6,
  "publication_date": "2026-05-25",
  "months_since_pub": 4,
  "citations_per_month": 1.5,
  "artifact_name": "AgingBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "maintain factual/behavioral reliability as the agent's effective state changes via compression, retrieval, revision, and maintenance over its deployment lifespan; correctly write, retrieve, and utilize memory across the memory pipeline's stages; correctly repair a diagnosed failure at the specific pipeline stage (write/retrieval/utilization) responsible for it",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "AgingBench organizes aging into four mechanisms (compression, interference, revision, maintenance aging) diagnosed via 'temporal dependency graphs and paired counterfactual probes,' meaning later behavior/factual state depends causally on the accumulated history of writes, retrievals, and revisions across many prior sessions.",
  "n_goals": "4 aging mechanisms (compression, interference, revision, maintenance), diagnosed at write/retrieval/utilization pipeline stages, across 7 scenarios and 14 models",
  "tracking_demand": "The agent (and the benchmark's diagnostic layer) must track the full lifespan history of memory writes, retrievals, and revisions across up to 200 sessions to determine which pipeline stage a given failure traces back to.",
  "scoring": "other:stage-targeted-diagnostic-profiling. The paper reports behavioral/factual-precision tests plus 'diagnostic profiles' pinpointing which pipeline stage a failure traces to -- process-level diagnosis rather than a single pass/fail score.",
  "horizon_value": "~400 runs spanning 8-200 sessions",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per run (each run consists of a chained sequence of 8 to 200 sessions)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "AgingBench organizes agent aging into four mechanisms: compression aging, interference aging, revision aging, and maintenance aging. To diagnose these failures, AgingBench uses temporal dependency graphs and paired counterfactual probes that produce diagnostic profiles for the write, retrieval, and utilization stages of the memory pipeline.",
  "horizon_span": "over ~400 runs spanning 8 - 200 sessions show that agent aging is not one-dimensional",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288671771",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "sessions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286961600",
  "title": "AlpsBench: An LLM Personalization Benchmark for Real-Dialogue Memorization and Preference Alignment",
  "year": 2026,
  "venue": "Annual International ACM SIGIR Conference on Research and Development in Information Retrieval",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 6,
  "publication_date": "2026-03-09",
  "months_since_pub": 6,
  "citations_per_month": 1.0,
  "artifact_name": "AlpsBench",
  "artifact_kind": "dataset",
  "domain": "personal-assistant-memory",
  "goal_types": "extract explicit and implicit personalized user traits from long-term interaction sequences; correctly update stored personalized memory as new information arrives; retrieve relevant personalized memory under large distractor pools; utilize retrieved memory to produce preference-aligned, emotionally resonant responses",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each memory-lifecycle stage depends on the fidelity of the previous one: retrieval accuracy depends on correct earlier updating, and response utilization quality depends on both prior extraction and retrieval succeeding.",
  "n_goals": "2,500 long-term interaction sequences; four pivotal tasks per lifecycle stage (extraction, updating, retrieval, utilization)",
  "tracking_demand": "Agent must extract, update, retrieve, and utilize structured user memories (explicit and implicit personalization signals) consistently across long-term interaction sequences curated from real dialogues.",
  "scoring": "other:per-task-accuracy-across-four-lifecycle-stages",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We define four pivotal tasks---personalized information extraction, updating, retrieval, and utilization ---and establish protocols to evaluate the entire lifecycle of memory management.",
  "horizon_span": "AlpsBench comprises 2,500 long-term interaction sequences curated from WildChat, paired with human-verified structured memories",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286961600",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284717806",
  "title": "PersonalAlign: Hierarchical Implicit Intent Alignment for Personalized GUI Agent with Long-Term User-Centric Records",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 16,
  "publication_date": "2026-01-14",
  "months_since_pub": 8,
  "citations_per_month": 2.0,
  "artifact_name": "AndroidIntent",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "resolve omitted preferences in vague GUI instructions using long-term user records; anticipate latent routines from user state for proactive assistance; correctly execute and proactively suggest actions grounded in hundreds of distinct user-specific preferences and routines",
  "goal_origin": "implied-by-constraints",
  "decomposition": "hierarchical",
  "interdependence": "Resolving a vague instruction correctly depends on having accurately inferred the relevant preference or routine from long-term records, and proactive suggestions depend on correctly anticipating latent routines tied to the current user state.",
  "n_goals": "775 annotated user-specific preferences and 215 routines drawn from 20k long-term records",
  "tracking_demand": "Agent must maintain a continuously updating personal memory, hierarchically organizing user preferences and routines inferred from long-term records, to resolve vague instructions and proactively suggest actions.",
  "scoring": "other:execution-and-proactive-performance-improvement",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "HIM-Agent improves both execution and proactive performance by 15.7% and 7.3% respectively over baseline GUI agents (GPT-5, Qwen3-VL, UI-TARS evaluated); no human baseline given.",
  "availability": null,
  "goal_span": "We annotated 775 user-specific preferences and 215 routines from 20k long-term records across different users for evaluation... HIM-Agent significantly improves both execution and proactive performance by 15.7% and 7.3%.",
  "horizon_span": "We annotated 775 user-specific preferences and 215 routines from 20k long-term records across different users for evaluation.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284717806",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288257069",
  "title": "CalBench: Evaluating Coordination-Privacy Trade-offs in Multi-Agent LLMs",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 4,
  "publication_date": "2026-05-10",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "CalBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "schedule a stream of M incoming meetings while managing one's own private calendar; minimize disruption cost to the agent's own calendar; coordinate with other agents' private calendars via language-mediated negotiation, without directly inspecting their calendars; preserve privacy (avoid over-revealing calendar information) while still achieving fair burden allocation across agents",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each of the N agents manages a private calendar so no one can inspect another's, meaning success on the incoming stream of M meetings requires ongoing language-mediated coordination, and accepting one meeting affects calendar capacity available for scheduling subsequent meetings in the stream.",
  "n_goals": "N agents scheduling a stream of M incoming meetings (specific N, M not given in the abstract)",
  "tracking_demand": "Each agent must track its own private calendar state, disruption costs incurred so far, what it has revealed or withheld to other agents, and burden/fairness considerations across the stream of scheduling requests.",
  "scoring": "other:multi-metric-task-success-cost-privacy - evaluates task success, excess cost, communication efficiency, burden fairness, and privacy leakage under matched information constraints; abstract explicitly notes 'completion alone misses important failures', implying multi-dimensional scoring rather than a simple binary pass/fail, though no formal per-meeting subgoal checkpoint is described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Across seven model families, agents leave avoidable cost on the table, communication volume does not predict lower regret, and privacy-preserving silence can deprive teammates of cost information needed for fair burden allocation; no single top-line number or human/expert baseline is given.",
  "availability": null,
  "goal_span": "In each task, $N$ agents manage separate private calendars and schedule a stream of $M$ incoming meetings while minimizing disruption costs. Because no agent can inspect another agent's calendar, success requires language-mediated coordination rather than centralized planning.",
  "horizon_span": "In each task, $N$ agents manage separate private calendars and schedule a stream of $M$ incoming meetings while minimizing disruption costs.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288257069",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284911920",
  "title": "PEARL: Self-Evolving Assistant for Time Management with Reinforcement Learning",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 3,
  "publication_date": "2026-01-17",
  "months_since_pub": 8,
  "citations_per_month": 0.38,
  "artifact_name": "CalConflictBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "resolve calendar conflicts round-by-round across a full calendar year; infer and progressively adapt to evolving user preferences (attendee priorities, topic importance, time/location preferences); decide which meetings to attend, reschedule, or decline per conflict",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Preferences inferred from earlier rounds of conflict resolution inform and constrain decisions in later rounds across the calendar year, so decision/ranking quality compounds based on how well the agent has adapted its preference model over time.",
  "n_goals": null,
  "tracking_demand": "Agent must maintain an external preference memory that stores and updates inferred strategies (attendee priorities, topic importance, time/location preferences) and use round-wise decisions across the calendar year to track scheduling state.",
  "scoring": "other:round-wise-error-rate",
  "horizon_value": "one calendar year (presented round-by-round)",
  "horizon_unit": "simulated-years",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (one calendar year of conflict-resolution rounds)",
  "horizon_stated": "yes",
  "headline_result": "PEARL achieves an error reduction rate of 0.76 and a 55% improvement in average error rate versus the strongest baseline; the strongest baseline, Qwen-3-30B-Think, has an average error rate of 35%; no human baseline given.",
  "availability": null,
  "goal_span": "In CalConflictBench, conflicts are presented to agents round-by-round over a calendar year, requiring them to infer and adapt to user preferences progressively... (ii) optimizes the agent with round-wise rewards that directly supervise decision correctness, ranking quality, and memory usage across rounds.",
  "horizon_span": "In CalConflictBench, conflicts are presented to agents round-by-round over a calendar year, requiring them to infer and adapt to user preferences progressively.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284911920",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-years",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288671242",
  "title": "Claw-Anything: Benchmarking Always-On Personal Assistants with Broader Access to User's Digital World",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-05-25",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "Claw-Anything",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "reason over long-horizon activity histories accumulated across months of simulated user activity; coordinate interdependent backend services and integrated GUI/CLI interaction across multiple devices; remain robust to irrelevant events and conflicting noise signals; proactively anticipate user needs and deliver timely recommendations",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "open-ended",
  "interdependence": "Backend services are interdependent, so actions and state changes accumulated across simulated months of activity (including injected noise and conflicting signals) constrain what later reasoning and proactive recommendations must correctly account for.",
  "n_goals": null,
  "tracking_demand": "Agent must reason over rich, long-horizon activity histories and interdependent backend-service state accumulated across simulated months, remaining robust to irrelevant/conflicting noise while proactively anticipating user needs.",
  "scoring": "other:pass@1-rate",
  "horizon_value": "months",
  "horizon_unit": "other:simulated-months",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (accumulated simulated user activity history)",
  "horizon_stated": "yes",
  "headline_result": "GPT-5.5 achieves only 34.5% pass@1, substantially below prior benchmarks; a released data-generation pipeline yielding 2,000 training environments improves the base model by 23.7%; no human baseline given.",
  "availability": null,
  "goal_span": "To instantiate this setting, we simulate months of user activity through multi-round event injection, producing complex world states and realistic noise, including irrelevant events and conflicting signals. Agents must reason over rich contextual environments while remaining robust to such noise.",
  "horizon_span": "we simulate months of user activity through multi-round event injection, producing complex world states and realistic noise, including irrelevant events and conflicting signals.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288671242",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284648447",
  "title": "CloneMem: Benchmarking Long-Term Memory for AI Clones",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 10,
  "publication_date": "2026-01-11",
  "months_since_pub": 8,
  "citations_per_month": 1.25,
  "artifact_name": "CloneMem",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "track an individual's evolving experiences, emotions, opinions, and personal states across long non-conversational digital traces (diaries, social posts, emails); answer/act using the current (not superseded) personal state reflecting one to three years of life history",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Personal states are updated over time by new diary/social/email entries, so later evaluation tasks depend on correctly integrating the temporally ordered stream of one to three years of digital traces rather than any single entry in isolation.",
  "n_goals": null,
  "tracking_demand": "Agent must track an evolving personal state (experiences, emotions, opinions) over one to three years of non-conversational digital traces and correctly distinguish current from superseded information.",
  "scoring": "other:longitudinal-state-tracking-tasks - defines tasks assessing an agent's ability to track evolving personal states; abstract frames this around correctly tracking state over time rather than a single final answer, but does not detail a formal per-checkpoint credit scheme.",
  "horizon_value": "1 to 3",
  "horizon_unit": "simulated-years",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole scenario (spans a protagonist's simulated life history of digital traces, not a single interaction turn)",
  "horizon_stated": "yes",
  "headline_result": "Current memory mechanisms struggle in this setting; no specific numeric top score is given in the abstract.",
  "availability": "https://github.com/AvatarMemory/CloneMemBench",
  "goal_span": "CloneMem adopts a hierarchical data construction framework to ensure longitudinal coherence and defines tasks that assess an agent's ability to track evolving personal states.",
  "horizon_span": "We introduce CloneMem, a benchmark for evaluating longterm memory in AI Clone scenarios grounded in non-conversational digital traces, including diaries, social media posts, and emails, spanning one to three years.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284648447",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-years",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290984478",
  "title": "Controlled Memory Interference in Continual LLM Agents",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-08-07",
  "months_since_pub": 1,
  "citations_per_month": 1.0,
  "artifact_name": "Controlled Memory Interference (CMI)",
  "artifact_kind": "dataset",
  "domain": "personal-assistant-memory",
  "goal_types": "update memory upon new experience while managing reinforcement, revision, or interference with existing memory states; distinguish valid memory updates from interference-inducing memories; maintain multiple simultaneously relevant memories differing in state, temporal validity, or authority; preserve continuity across sessions to personalize behavior via accumulated experience",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "New experiences can reinforce, revise, or interfere with existing memory states that remain simultaneously relevant, so a later memory update's correctness depends on correctly distinguishing it from earlier, potentially conflicting or authoritative memories; poisoning is shown to be sensitive to update-authority cues rather than recency alone.",
  "n_goals": null,
  "tracking_demand": "The agent's memory system must track multiple simultaneously relevant memory states that differ in validity, temporal recency, or authority, and correctly determine which prior memory a new experience should reinforce, revise, or be blocked by.",
  "scoring": "other:update-plasticity-vs-stability-tradeoff. The framework measures how relationship-specific interference affects update plasticity and stability (benign accumulation has limited effects, whereas interference sharply suppresses plasticity with little stability gain) rather than a single binary success score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce Controlled Memory Interference (CMI), a controlled diagnostic and data-generation framework for studying how agent memory evolves under different memory relationships. Across controlled memory evolution, benign accumulation has limited effects, whereas relationship-specific interference sharply suppresses update plasticity with little stability gain, either by blocking target-memory exposure or by disrupting its downstream use.",
  "horizon_span": "Long-term memory enables AI agents to maintain continuity across sessions, personalize behavior, and evolve through accumulated experience.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290984478",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285269543",
  "title": "ES-MemEval: Benchmarking Conversational Agents on Personalized Long-Term Emotional Support",
  "year": 2026,
  "venue": "The Web Conference",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 11,
  "publication_date": "2026-02-02",
  "months_since_pub": 7,
  "citations_per_month": 1.57,
  "artifact_name": "ES-MemEval / EvoEmo",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "extract and retain implicit, fragmented user disclosures across sessions; perform temporal reasoning over how the user's state has evolved; detect conflicts between what the user said earlier and later; abstain when information is insufficient rather than hallucinate; build and update a model of the user across QA, summarization, and dialogue-generation tasks",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Later-session answers depend on correctly integrating dispersed, sometimes conflicting, information disclosed by the user in earlier sessions, so failing to track evolving user state corrupts later responses.",
  "n_goals": "5 (core memory capabilities: information extraction, temporal reasoning, conflict detection, abstention, user modeling)",
  "tracking_demand": "Agent must track fragmented and implicit user disclosures, detect when the user's state or facts have changed, and maintain an updated user model across multiple sessions of an emotional-support dialogue.",
  "scoring": "other:not-stated \u2014 the abstract describes evaluating five separate memory capabilities but does not specify whether scoring aggregates into one score or gives capability-level partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Explicit long-term memory reduces hallucinations and improves personalization versus baselines, per the authors; no single best-model score or human-expert gap is given.",
  "availability": null,
  "goal_span": "we introduce ES-MemEval, a comprehensive benchmark that systematically evaluates five core memory capabilities\u2014information extraction, temporal reasoning, conflict detection, abstention, and user modeling\u2014in long-term emotional support scenarios",
  "horizon_span": "we also propose EvoEmo, the first multi-session dataset for personalized long-term emotional support scenarios, capturing fragmented, implicit user disclosures and evolving user states",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285269543",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "290833099",
  "title": "EduClaw-Bench: A Long-Horizon Benchmark for Pedagogical LLM Agents with Simulated Learners",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-08-04",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "EduClaw-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "improve a simulated learner's knowledge-concept mastery over a sustained tutoring relationship; personalize responsiveness and helpfulness to the learner's evolving needs; apply sound curriculum-design principles (Gagne and Rosenshine axes) across the relationship; sustain good tutoring performance across the full 30-day horizon, not just in an initial session",
  "goal_origin": "mixed:given-up-front-relationship-structure-plus-emitted-by-environment-learner-state",
  "decomposition": "sequential-chain",
  "interdependence": "The simulated learner's knowledge-concept mastery evolves via a knowledge-tracing (KT) model driven by the tutor's own prior interactions, so a tutor's teaching choices earlier in the 30-day relationship causally shape the learner's state in later sessions; the paper explicitly finds that 'almost no combination sustains good tutoring over the full horizon.'",
  "n_goals": "55 scenarios; continuous 30-day relationship per scenario; 5 scoring axes (learning gain, responsiveness, helpfulness, Gagne, Rosenshine); 10 agent adapters over 3 base-model tiers",
  "tracking_demand": "The tutor agent must track the simulated learner's evolving knowledge-concept mastery (grounded in a KT model) across the 30-day relationship, adapting its teaching to what the learner has and hasn't yet learned, and sustaining quality across the full horizon rather than only an initial session.",
  "scoring": "other:multi-axis-judged-scoring. Each agent is 'scored on three primary axes (learning gain, responsiveness, and helpfulness) and two curriculum-design axes (Gagne and Rosenshine),' with the latter judged by 'a cross-family panel of three LLM judges' -- explicit multi-dimensional scoring, though not framed as a single subgoal-checkpoint/milestone rubric.",
  "horizon_value": "continuous 30-day tutoring relationship, across 55 scenarios",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (one whole 30-day tutor-learner relationship per scenario)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce EduClaw-Bench, a benchmark that places an agent tutor in a continuous 30-day relationship with a simulated learner grounded in knowledge tracing (KT), whose knowledge-concept mastery, from a KT model trained on real-student data, drives its answers and is probed for learning gain across 55 scenarios.",
  "horizon_span": "We introduce EduClaw-Bench, a benchmark that places an agent tutor in a continuous 30-day relationship with a simulated learner",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290833099",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288259502",
  "title": "EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 3,
  "publication_date": "2026-05-11",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "EgoMemReason",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "track how object states evolve and change across days (entity memory); recall and correctly order activities separated by hours or days (event memory); abstract recurring patterns from sparse, repeated observations across a whole week (behavior memory); answer each of 500 questions requiring integration of evidence across multiple days of egocentric video",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each question requires integrating an average of 5.1 video-segment pieces of evidence spread across up to 25.9 hours of backtracked memory, so evidence gathered at different points in the week must be jointly reconciled (e.g., correct temporal ordering, tracking the same entity's evolving state) to answer correctly.",
  "n_goals": "500 questions across three memory types and six core challenges; average 5.1 evidence segments per question",
  "tracking_demand": "The system must accumulate information over an entire week of continuous egocentric video, recall prior states, track the temporal order of events separated by hours or days, and abstract recurring behavioral patterns from sparse repeated observations, backtracking an average of 25.9 hours of memory per question.",
  "scoring": "other:accuracy-per-memory-type-and-challenge-category",
  "horizon_value": "25.9 hours of memory backtracking per question (average); week-long underlying video",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per question, within a week-long continuous video)",
  "horizon_stated": "yes",
  "headline_result": "Even the best of 17 evaluated methods (MLLMs and agentic frameworks) achieves only 39.6% overall accuracy, with performance degrading as evidence spans longer temporal horizons; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "EgoMemReason comprises 500 questions across three memory types and six core challenges, with an average of 5.1 video segments of evidence per question and 25.9 hours of memory backtracking.",
  "horizon_span": "EgoMemReason comprises 500 questions across three memory types and six core challenges, with an average of 5.1 video segments of evidence per question and 25.9 hours of memory backtracking.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288259502",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285271100",
  "title": "Evaluating Long-Horizon Memory for Multi-Party Collaborative Dialogues",
  "year": 2026,
  "venue": "Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 11,
  "publication_date": "2026-02-01",
  "months_since_pub": 7,
  "citations_per_month": 1.57,
  "artifact_name": "EverMemBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "perform fine-grained recall across dense, cross-topic multi-party conversations; maintain memory awareness of implicitly relevant information beyond similarity retrieval; understand and correctly attribute user profiles across multiple participants and roles; resolve multi-hop reasoning under multi-party attribution and temporally evolving decisions",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Fine-grained recall, memory awareness, and user-profile understanding are evaluated over the same underlying multi-party, multi-group conversation history, so correctly attributing a fact to the right participant/role and time depends on correctly tracking evolving decisions across the whole conversation.",
  "n_goals": "2,400 QA pairs across three evaluation dimensions",
  "tracking_demand": "The agent must track role-conditioned personas, temporally evolving decisions, and cross-topic interleaved information across multi-party, multi-group conversations exceeding one million tokens, in order to answer QA pairs spanning recall, awareness, and profile understanding.",
  "scoring": "other:accuracy-by-dimension-recall-awareness-profile",
  "horizon_value": ">1,000,000",
  "horizon_unit": "other:tokens",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole multi-party conversation",
  "horizon_stated": "yes",
  "headline_result": "Multi-hop reasoning collapses under multi-party attribution even with oracle evidence (26% accuracy); no explicit human baseline given.",
  "availability": "https://github.com/EverMind-AI/EverMemBench",
  "goal_span": "EverMemBench evaluates memory systems using 2,400 QA pairs across three dimensions essential for real applications: fine-grained recall, memory awareness, and user profile understanding. Our evaluation reveals fundamental limitations of current systems: multi-hop reasoning collapses under multi-party attribution even with oracle evidence (26% accuracy), temporal reasoning fails without explicit version semantics beyond timestamps",
  "horizon_span": "built from multi-party, multi-group conversations spanning over one million tokens with dense cross-topic interleaving, temporally evolving decisions, and role-conditioned personas",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285271100",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288655681",
  "title": "EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 7,
  "publication_date": "2026-05-18",
  "months_since_pub": 4,
  "citations_per_month": 1.75,
  "artifact_name": "EvoMemBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "retain and retrieve knowledge-oriented information across episode boundaries; retain and reuse execution-oriented (procedural) experience across episode boundaries; satisfy in-episode memory demands; satisfy cross-episode memory demands",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Memory scope (in-episode vs. cross-episode) and memory content (knowledge- vs. execution-oriented) are evaluated as a 2x2 axis, and cross-episode conditions require that information stored in earlier episodes be correctly carried forward and applied in later ones.",
  "n_goals": null,
  "tracking_demand": "The system must store, update, and retrieve both knowledge and procedural experience across episode boundaries, and determine which stored memories are relevant to reuse for the current task.",
  "scoring": "other:comparative-performance-across-15-memory-methods-vs-long-context-baselines",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "No single method dominates: long-context baselines remain highly competitive and no memory form works consistently across all settings, so no clear top-model/human-gap figure is reported.",
  "availability": "https://github.com/DSAIL-Memory/EvoMemBench",
  "goal_span": "we introduce EvoMemBench, a unified benchmark organized along two axes: memory scope (in-episode vs. cross-episode) and memory content (knowledge-oriented vs. execution-oriented). We compare 15 representative memory methods with strong long-context baselines under a standardized protocol.",
  "horizon_span": "memory scope (in-episode vs. cross-episode) and memory content (knowledge-oriented vs. execution-oriented)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288655681",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "284532760",
  "title": "EvolMem: A Cognitive-Driven Benchmark for Multi-Session Dialogue Memory",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 4,
  "publication_date": "2026-01-07",
  "months_since_pub": 8,
  "citations_per_month": 0.5,
  "artifact_name": "EvolMem",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "correctly recall declarative memory content across multiple dialogue sessions; correctly exhibit non-declarative memory capabilities across multiple dialogue sessions; succeed across multiple fine-grained memory-ability dimensions grounded in cognitive psychology, not just one aggregate score",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "The benchmark decomposes memory into 'multiple fine-grained abilities' under declarative and non-declarative categories, tested across multi-session conversations of 'controllable complexity', where later-session queries depend on content and narrative structure established (topic-initiated, narrative-transformed) in earlier sessions.",
  "n_goals": null,
  "tracking_demand": "Agent/memory system must retain and correctly apply both declarative (fact-like) and non-declarative (procedural/implicit) memory content across multiple sessions of scalable, controllable complexity.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports that 'no LLM consistently outperforms others across all memory dimensions' and that agent memory mechanisms 'often exhibit notable efficiency limitations', implying per-dimension (fine-grained) scoring rather than one aggregate pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "No LLM consistently outperforms others across all memory dimensions; agent memory mechanisms do not necessarily enhance LLMs' capabilities and often show efficiency limitations (no single numeric headline score given).",
  "availability": "https://github.com/shenye7436/EvolMem",
  "goal_span": "EvolMem is grounded in cognitive psychology and encompasses both declarative and non-declarative memory, further decomposed into multiple fine-grained abilities.",
  "horizon_span": "This framework enables scalable generation of multi-session conversations with controllable complexity, accompanied by sample-specific evaluation guidelines.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284532760",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "287203118",
  "title": "FileGram: Grounding Agent Personalization in File-System Behavioral Traces",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 4,
  "publication_date": "2026-04-06",
  "months_since_pub": 5,
  "citations_per_month": 0.8,
  "artifact_name": "FileGramBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "reconstruct an evolving user profile from dense file-system behavioral traces; disentangle overlapping/interleaved behavioral traces belonging to different activities; detect persona drift as the user's behavior changes over time; correctly ground multimodal (procedural, semantic, episodic) evidence into the user profile",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "FileGramOS builds user profiles directly from atomic actions and content deltas rather than dialogue summaries, encoding traces into procedural, semantic, and episodic channels with query-time abstraction, so profile reconstruction and drift detection both depend on correctly disentangling which atomic action belongs to which behavioral trace over time.",
  "n_goals": "FileGramEngine produces 640 controlled trajectories across 6 behavioral dimensions and 20 user profiles; FileGramBench has four tracks",
  "tracking_demand": "The memory system must track atomic file-system actions and content deltas over time, encode them into procedural/semantic/episodic channels, disentangle overlapping traces, and detect when the user's persona has drifted from its established profile.",
  "scoring": "other:per-track-diagnostic-performance-profile-reconstruction-disentanglement-drift-multimodal-grounding",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "FileGramBench remains challenging for state-of-the-art memory systems; FileGramEngine and FileGramOS are shown to be effective; no single numeric headline figure given in the abstract.",
  "availability": "https://github.com/synvo-ai/FileGram",
  "goal_span": "we propose FileGram, a comprehensive framework that grounds agent memory and personalization in file-system behavioral traces, comprising three core components: (1) FileGramEngine, a scalable persona-driven data engine ...; (2) FileGramBench, a diagnostic benchmark grounded in file-system behavioral traces for evaluating memory systems on profile reconstruction, trace disentanglement, persona drift detection, and multimodal grounding; and (3) FileGramOS, a bottom-up memory architecture",
  "horizon_span": "FileGramEngine produces 640 controlled trajectories with ground-truth labels across 6 behavioral dimensions and 20 user profiles",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287203118",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285540833",
  "title": "Gaia2: Benchmarking LLM Agents on Dynamic and Asynchronous Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 21,
  "publication_date": "2026-02-12",
  "months_since_pub": 7,
  "citations_per_month": 3.0,
  "artifact_name": "Gaia2 / Agents Research Environments (ARE)",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "complete a scenario-level task while the environment evolves independently of the agent's actions; operate under explicit temporal constraints (time-sensitive tasks); adapt to noisy and dynamic events injected during the scenario; resolve ambiguity in requests; collaborate with other agents present in the scenario",
  "goal_origin": "mixed:given-up-front-plus-emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Because the environment evolves asynchronously (independent of agent actions), a write-action verifier checks fine-grained, action-level correctness; time-sensitive tasks mean an action taken too late invalidates a goal even if the action itself is correct, coupling goal success to the timing of other concurrent scenario events.",
  "n_goals": "1,120 human-annotated scenarios (per escalated search)",
  "tracking_demand": "The agent must track evolving environment state (independent of its own actions), remaining time budgets for time-sensitive sub-goals, ambiguity-resolution status, and coordination with other agents, verified at the action level by write-action verifiers.",
  "scoring": "other:pass-at-1-with-action-level-write-verifiers",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "GPT-5 (high) reaches the strongest overall score at 42% pass@1 but fails on time-sensitive tasks; Claude-4 Sonnet trades accuracy and speed for cost; Kimi-K2 leads open-source models at 21% pass@1; no human baseline given.",
  "availability": null,
  "goal_span": "Gaia2 introduces scenarios where environments evolve independently of agent actions, requiring agents to operate under temporal constraints, adapt to noisy and dynamic events, resolve ambiguity, and collaborate with other agents. Each scenario is paired with a write-action verifier, enabling fine-grained, action-level evaluation and making Gaia2 directly usable for reinforcement learning from verifiable rewards.",
  "horizon_span": "Gaia2 consists of 1,120 human-annotated scenarios set in a smartphone-like environment with realistic apps (email, messaging, calendar, contacts, etc.)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285540833",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289625916",
  "title": "GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 3,
  "publication_date": "2026-06-17",
  "months_since_pub": 3,
  "citations_per_month": 1.0,
  "artifact_name": "GateMem",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "serve legitimate long-horizon requests that require state updates to shared memory; enforce access control across contextual authorization boundaries for different principals; perform agent-facing active forgetting after explicit deletion requests; avoid leaking unauthorized or deleted information to any principal",
  "goal_origin": "mixed:given-up-front-plus-injected-by-user-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "Multiple principals write to and query a common memory pool under different roles/scopes, so satisfying one principal's access request must not violate another principal's authorization boundary or a prior deletion request; utility, access-control, and forgetting goals trade off against one another.",
  "n_goals": null,
  "tracking_demand": "The agent must track per-principal roles, scopes and relationships, incremental memory updates, hidden checkpoints, and outstanding deletion requests across long-form multi-party episodes averaging roughly 200-240 turns depending on domain.",
  "scoring": "other:joint-utility-access-control-and-forgetting-scoring",
  "horizon_value": "~204.5 (medical) / 241.2 (office) / 224.9 (education) / 224.0 (household) turns per episode",
  "horizon_unit": "turns",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per episode, averaged by institutional domain",
  "horizon_stated": "yes",
  "headline_result": "No method simultaneously achieves strong utility, robust access control, and reliable forgetting; long-context prompting yields the best governance score at high token cost, while retrieval/external-memory methods reduce cost but still leak information.",
  "availability": null,
  "goal_span": "GateMem jointly evaluates utility for legitimate long-horizon requests with state updates, access control across contextual authorization boundaries, and agent-facing active forgetting after explicit deletion requests. It spans medical, office, education, and household domains, with long-form multi-party episodes, incremental memory injection, hidden checkpoints, structured judging, and leak-target annotations.",
  "horizon_span": "Medical: 204.5 ... Office: 241.2 ... Education: 224.9 ... Household: 224.0 (turns per episode, Table 2, dataset overview)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289625916",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "turns",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285659353",
  "title": "Multi-Agent Home Energy Management Assistant",
  "year": 2026,
  "venue": "SoftwareX",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 7,
  "publication_date": "2026-02-16",
  "months_since_pub": 7,
  "citations_per_month": 1.0,
  "artifact_name": "HEMA (Home Energy Management Assistant)",
  "artifact_kind": "other:deployed-multi-agent-assistant-system",
  "domain": "personal-assistant-memory",
  "goal_types": "sustain multi-turn conversational collaboration with preserved context across a home-energy-management session; perform energy consumption analysis and cost optimization (Analysis agent); answer educational queries and provide rebate information (Knowledge agent); control and schedule smart devices (Control agent); correctly route each user query to the right specialized agent via a self-consistency classifier",
  "goal_origin": "given-up-front",
  "decomposition": "other:three-specialized-agents-coordinated-by-a-routing-classifier",
  "interdependence": "The routing classifier must correctly dispatch each query to the Analysis, Knowledge, or Control agent, and because context is preserved across turns, later routing/response decisions depend on state and context accumulated from earlier turns in the same sustained conversation.",
  "n_goals": "three specialized agents (Analysis, Knowledge, Control); 36 purpose-built domain-specific tools; 23 objective evaluation metrics",
  "tracking_demand": "The system must preserve conversational context across multiple turns of sustained human-AI collaboration, track which of the three specialized agents/tools have been invoked, and maintain consistency across energy analysis, educational, and device-control interactions within one session.",
  "scoring": "other:LLM-as-simulated-user-with-23-objective-metrics",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "HEMA also includes a comprehensive evaluation framework using an LLM-as-simulated-user methodology with 23 objective metrics across task performance, factual accuracy, interaction quality, and system efficiency, allowing systematic testing across diverse scenarios and user personas without requiring extensive human subject testing.",
  "horizon_span": "multi-turn conversational interactions with preserved context",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": "Published in SoftwareX (a software-description venue) as a deployed assistant system with an accompanying evaluation framework, rather than a benchmark/dataset primarily intended for cross-system comparison; may not belong in a corpus of long-horizon agent benchmarks.",
  "url": "https://api.semanticscholar.org/CorpusId:285659353",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287432975",
  "title": "Evaluating Memory Capability in Continuous Lifelog Scenario",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 2,
  "publication_date": "2026-04-13",
  "months_since_pub": 5,
  "citations_per_month": 0.4,
  "artifact_name": "LifeDialBench (EgoMem / LifeMem)",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "recall and reason correctly over continuously accumulating lifelog conversation history; answer queries using only information available up to the query time (no temporal leakage) under an Online Evaluation protocol",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Because the Online Evaluation protocol 'strictly adheres to temporal causality', every answer may only depend on information disclosed earlier in the continuous stream, so later queries are constrained by, and must correctly integrate, everything the system has previously ingested.",
  "n_goals": null,
  "tracking_demand": "System must ingest and retain a continuously growing stream of ambient conversation (lifelog audio) and answer later queries using only causally-prior information, without being able to look ahead.",
  "scoring": "other:not-stated \u2014 the abstract reports a comparative finding (sophisticated memory systems fail to beat a simple RAG baseline) but does not describe a partial-credit or checkpoint scoring scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Current sophisticated memory systems fail to outperform a simple RAG-based baseline (no specific numeric score given in the abstract).",
  "availability": null,
  "goal_span": "we propose an \\textbf{Online Evaluation} protocol that strictly adheres to temporal causality, ensuring systems are evaluated in a realistic streaming fashion",
  "horizon_span": "wearable devices can continuously lifelog ambient conversations, creating substantial opportunities for memory systems",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287432975",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "288940457",
  "title": "LifeSide: Benchmarking Agents as Lifelong Digital Companions",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-06-03",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "LifeSide",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "integrate cross-session memory cues about a persistent user persona; continually update the agent's understanding of the user over time; adapt to the user's shifting privacy boundaries; sustain accurate emotional companionship across sessions",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "Cross-session Memory-Emotion-Environment loops mean later-session understanding, privacy decisions, and companionship responses all depend on correctly retaining and updating state accumulated in earlier sessions of the same persona.",
  "n_goals": "2,000 personas and 111K tasks; averaging 56.79 sessions per persona",
  "tracking_demand": "The agent must track a persistent user world (layered profile, event trajectory) across an average of 56.79 sessions and 851.85 user turns per persona, covering memory tracking, user understanding, privacy control, and emotional companionship.",
  "scoring": "other:per-dimension-scores-across-memory-understanding-privacy-companionship",
  "horizon_value": "avg 56.79 sessions / 851.85 user turns / 29.61K dialogue tokens per persona",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per persona (aggregated across that persona's full multi-session history)",
  "horizon_stated": "yes",
  "headline_result": "Even models that saturate current memory benchmarks fail to sustain accurate user understanding and true companionship over long horizons; no single numeric headline given in the abstract.",
  "availability": null,
  "goal_span": "\\benchmark uses multi-agent simulation to project environmental dynamics into dialogue, preserving the critical gap between latent thoughts and observable expressions. Evaluating 2,000 personas and 111K tasks across memory tracking, user understanding, privacy control, and emotional companionship, our experiment results reveal a stark reality: even models that saturate current memory benchmarks fail to sustain accurate user understanding and true companionship over long horizons.",
  "horizon_span": "This longitudinal design yields substantial interaction histories, averaging 56.79 sessions, 851.85 user turns, and 29.61K visible dialogue tokens per persona.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288940457",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "sessions",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286493352",
  "title": "LifeSim: Long-Horizon User Life Simulator for Personalized Assistant Evaluation",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 10,
  "publication_date": "2026-03-12",
  "months_since_pub": 6,
  "citations_per_month": 1.67,
  "artifact_name": "LifeSim-Eval",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "complete the user's explicit intentions correctly; recognize and satisfy the user's implicit intentions; recover an accurate model of the user's evolving profile/preferences over the course of long-horizon assistance; produce high-quality responses across 8 life domains and 1,200 diverse scenarios",
  "goal_origin": "mixed:given-up-front-life-domain-scenarios-with-user-intentions-evolving-over-the-simulation",
  "decomposition": "hierarchical",
  "interdependence": "LifeSim's user simulator models cognition via the Belief-Desire-Intention (BDI) model for 'coherent life trajectories', so the user's later intentions and profile depend on the intention-driven behavior generated earlier in the same life trajectory, and the assistant must recover this evolving profile to succeed under the 'long-horizon' condition.",
  "n_goals": "1,200 diverse scenarios across 8 life domains",
  "tracking_demand": "Agent must recover and continuously update the user's profile (explicit and implicit intentions, preferences) as it evolves across a long-horizon, multi-scenario life trajectory, using a multi-turn interactive assessment method.",
  "scoring": "other:not-stated precisely \u2014 the abstract describes assessing 'abilities to complete explicit and implicit intentions, recover user profiles, and produce high-quality responses' as separate capabilities under both single-scenario and long-horizon settings, implying multi-dimensional (not single binary) scoring.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Current LLMs 'face significant limitations in handling implicit intention and long-term user preference modeling' under long-horizon settings (no specific numeric score given in the abstract).",
  "availability": null,
  "goal_span": "LifeSim-Eval covers 8 life domains and 1,200 diverse scenarios, and adopts a multi-turn interactive method to assess models'abilities to complete explicit and implicit intentions, recover user profiles, and produce high-quality responses.",
  "horizon_span": "LifeSim-Eval covers 8 life domains and 1,200 diverse scenarios",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286493352",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "287479488",
  "title": "LiveClawBench: Benchmarking LLM Agents on Complex, Real-World Assistant Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 5,
  "publication_date": "2026-03-20",
  "months_since_pub": 6,
  "citations_per_month": 0.83,
  "artifact_name": "LiveClawBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "resolve tasks that span cross-service dependencies across mocked applications; operate correctly despite contaminated/inconsistent prior state; correctly infer implicit user intent not explicitly stated in the request; adapt to runtime changes within stateful mock services during task execution",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are constructed via a Triple-Axis Complexity Framework (environment complexity, cognitive demand, runtime adaptability), so satisfying a cross-service task requires state consistency across dependent mocked services; contaminated state or an unresolved implicit intent in one service can block a downstream step in another.",
  "n_goals": "134 executable cases across 10 domains with 22 mocked services",
  "tracking_demand": "The agent must track session state, artifacts, and prior side-effects across 22 stateful mocked services, resolving cross-service dependencies and implicit intent while adapting to runtime changes during the task.",
  "scoring": "other:average-score-across-models",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Kimi-K2.7-Code achieves the best overall performance with an average score of 76.0, but even top models remain far from saturating the benchmark, especially on the Hard subset; no human baseline given.",
  "availability": null,
  "goal_span": "LiveClawBench combines a Triple-Axis Complexity Framework for difficulty-driven task construction with reproducible full-stack mock applications that preserve stateful execution semantics. With 134 executable cases across 10 domains with 22 mocked services, LiveClawBench supports controlled, extensible, and factor-level diagnostic evaluation of realistic agentic tasks.",
  "horizon_span": "The scatter shows that higher reward is not simply a consequence of taking more interaction steps.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287479488",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289625101",
  "title": "MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 2,
  "publication_date": "2026-06-23",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "MEMPROBE",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "assist simulated users across a trajectory of leak-controlled tasks while accumulating memory; recover/reconstruct a hidden, taxonomy-anchored user-state bank (31 dimensions) from the agent's own resulting memory; balance successful task assistance against auditable, faithful memory recovery",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The post-interaction memory artifact accumulates across the trajectory of leak-controlled tasks, and recoverability of each of the 31 hidden user-state dimensions depends on what was retained (and not lost) across earlier tasks in the same trajectory.",
  "n_goals": "50 simulated users with 31 hidden dimensions each (1,550 recovery targets)",
  "tracking_demand": "Agent must accumulate and retain a faithful memory of 31 hidden user-state dimensions across a trajectory of leak-controlled assistance tasks, since the memory is later audited by reconstructing the user-state bank from it under full-store and top-k access.",
  "scoring": "other:category-balanced-recovery-score",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Task completion nearly saturates even for a memoryless baseline, while category-balanced memory recovery stays moderate (about 0.6) and drops further under top-k retrieval, across 5 tested memory systems; no human baseline given.",
  "availability": null,
  "goal_span": "We instantiate this view in MEMPROBE, a benchmark in which a memory-equipped agent assists simulated users, each carrying a hidden, taxonomy-anchored user-state bank, across a trajectory of leak-controlled tasks, after which that bank is reconstructed from the agent's resulting memory under both full-store and top-k access... MEMPROBE spans 50 simulated users with 31 hidden dimensions each (1,550 recovery targets)",
  "horizon_span": "assists simulated users, each carrying a hidden, taxonomy-anchored user-state bank, across a trajectory of leak-controlled tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289625101",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291832154",
  "title": "When Does Memory Help? A Cost-Aware Evaluation of Long-Term Memory in Tool-Using LLM Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-07-26",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "MERIT",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "correctly recall and use an earlier-episode fact when executing a later, dependent tool-use task; correctly recall and use an UPDATED fact (superseding a stale one) rather than acting on outdated information; operate under an explicit cost budget (token/dollar metering) while doing so; avoid corrupted/adversarially degraded memory leading to incorrect actions",
  "goal_origin": "mixed:given-up-front-task-structure-plus-emitted-by-environment-fact-updates",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Episodes are grouped into arcs of 4-6 episodes sharing entities, with dependent-task success verified by an automated leak check, so a later episode's task explicitly depends on facts (including updated facts) established in an earlier episode of the same arc.",
  "n_goals": "23,440 scored episodes across a 3-model x 3-seed grid plus a two-generation pilot; episodes grouped into arcs of 4-6 episodes sharing entities (10 arcs x 5 episodes per domain x difficulty x condition)",
  "tracking_demand": "The agent's memory system must retain facts (including corrections to previously stored facts) across an arc of linked episodes and correctly retrieve and act on the current, updated version of a fact rather than a stale cached one, all while the harness meters the token/dollar cost of every memory operation.",
  "scoring": "other:dependent-task-success-with-cost-accounting. Memory 'lifts dependent-task success from a leak-verified floor of 0.00 to 0.55-1.00,' separating a no-memory floor from memory-enabled task success and further reporting cost-normalized 'marginal utility per dollar,' a continuous, cost-aware outcome measure rather than a subgoal-checkpoint rubric.",
  "horizon_value": "episodes grouped into arcs of 4-6 linked episodes sharing entities; 10 arcs x 5 episodes per (domain x difficulty x condition), 23,440 scored episodes total",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "across an arc (multiple linked episodes, not a single episode) -- dependent-task success is measured across this cross-episode chain",
  "horizon_stated": "yes",
  "headline_result": "Swapping a memory implementation moves task success by up to 60 percentage points, and full replay (the most expensive strategy) is never economical: the best condition per domain delivers 2.7-3.9x its marginal utility per dollar versus full replay; no human/expert baseline given.",
  "availability": null,
  "goal_span": "MERIT provides episodic tool-use tasks in three domains whose dependence on earlier-episode facts is verified by an automated leak check; a difficulty ladder ending in updated-fact recall; controlled memory corruption; and full token and dollar metering of every memory operation... memory lifts dependent-task success from a leak-verified floor of 0.00 to 0.55-1.00. On updated facts, embedding retrieval collapses unpredictably (0.30-0.95 across models; max seed gap 0.45), and agents act on a correctly retrieved value only 55% of the time.",
  "horizon_span": "Episodes are grouped into arcs of 4-6 episodes sharing entities... 10 arcs x 5 episodes per (domain x difficulty x condition)",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291832154",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288655900",
  "title": "MemConflict: Evaluating Long-Term Memory Systems Under Memory Conflicts",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 5,
  "publication_date": "2026-05-20",
  "months_since_pub": 4,
  "citations_per_month": 1.25,
  "artifact_name": "MemConflict",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "retrieve and rank the temporally valid, factually correct, and contextually applicable memory candidate when multiple conflicting alternatives exist; correctly answer queries under dynamic, static, and conditional conflict types despite distractors and long conflict distances",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Later queries depend on correctly resolving which of several temporally- or contextually-conflicting earlier memory statements remains valid, with conflict distances spanning many sessions (dynamic: 5-25; static: 10-45; conditional: 9-49), so retrieval/ranking must reconcile information injected far apart in the history.",
  "n_goals": "average of 52.33 sessions and 2,349.17 dialogue turns per benchmark instance (across 12 virtual users), with conflict distances ranging from 5 to 49 sessions depending on conflict type",
  "tracking_demand": "The agent's memory system must retrieve and rank memory candidates while tracking temporal validity, factual correctness, and contextual applicability across an average of 52.33 sessions (2,349.17 turns, ~203,910 tokens) per instance, correctly resolving conflicts placed 5 to 49 sessions apart.",
  "scoring": "other:black-box-plus-white-box -- the multi-session dialogue benchmark supports black-box evaluation of final answers and white-box analysis of supporting-memory retrieval and ranking, i.e. both outcome-level and process-level credit rather than a single final-answer score alone.",
  "horizon_value": "average of 52.33 sessions and 2,349.17 dialogue turns (about 203,910.83 tokens of context) per benchmark instance; conflict distances span 5-25 sessions (dynamic), 10-45 sessions (static), and 9-49 sessions (conditional)",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per instance/episode (one simulated user's full long-horizon history)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "MemConflict formalizes dynamic, static, and conditional conflicts over temporal validity, factual correctness, and contextual applicability. It simulates controlled long-horizon histories from structured user profiles, introduces cross-session conflicts, and injects semantically similar distractors to create competition among memory candidates. The resulting multi-session dialogue benchmark supports black-box evaluation of final answers and white-box analysis of supporting-memory retrieval and ranking.",
  "horizon_span": "Average Session Number: 52.33; Average Dialogue Turns: 2,349.17; Average Context Length (tokens): 203,910.83",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288655900",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "sessions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287508647",
  "title": "MemGround: Long-Term Memory Evaluation Kit for Large Language Models in Gamified Scenarios",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-03-23",
  "months_since_pub": 6,
  "citations_per_month": 0.17,
  "artifact_name": "MemGround",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "recall surface-level game state facts (Surface State Memory); associate events across time (Temporal Associative Memory); perform reasoning that depends on accumulated memory (Reasoning-Based Memory); unlock/discover memory fragments in the correct order across a gamified scenario",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "The three memory tiers build on one another (surface recall feeds temporal association, which feeds reasoning), and correct-order unlocking of memory fragments is explicitly scored, so out-of-order or missed fragments break downstream reasoning tasks.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain and update a three-tier memory store (surface state, temporal associations, reasoning-derived facts) across continuous gamified interactions, tracked via Memory Fragments Unlocked and Memory Fragments with Correct Order metrics, with runs capped at 600-1000 interaction steps depending on task type.",
  "scoring": "milestone-rubric",
  "horizon_value": "max 600-1000 interaction steps (task-dependent); early stop after 200 consecutive steps with no new discovery",
  "horizon_unit": "other:interaction-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/run",
  "horizon_stated": "yes",
  "headline_result": "State-of-the-art LLMs and memory agents still struggle with sustained dynamic tracking, temporal event association, and complex reasoning derived from long-term accumulated evidence; no explicit human baseline given.",
  "availability": null,
  "goal_span": "MemGround introduces a three-tier hierarchical framework that evaluates Surface State Memory, Temporal Associative Memory, and Reasoning-Based Memory through specialized interactive tasks. ... a multi-dimensional metric suite comprising Question-Answer Score (QA Overall), Memory Fragments Unlocked (MFU), Memory Fragments with Correct Order (MFCO), and Exploration Trajectory Diagrams (ETD).",
  "horizon_span": "we set the maximum number of interaction steps to 600. ... we set the maximum number of interaction steps to 1000. ... if the model does not discover any new files within 200 consecutive steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287508647",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290178380",
  "title": "MemOps: Benchmarking Lifecycle Memory Operations in Long-Horizon Conversations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 3,
  "publication_date": "2026-07-14",
  "months_since_pub": 2,
  "citations_per_month": 1.5,
  "artifact_name": "MemOps",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "correctly execute each lifecycle memory operation (remember, forget, update, reflect, and their compositions) at the right point in a conversation; maintain a consistent, ordered memory-state trajectory (not just the correct final answer) across a long conversation",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later memory operations (e.g. an update or reflection) depend on the state established by earlier operations on the same fact/target, so applying an operation to a stale or wrong prior state produces an incorrect memory trajectory even if the final answer coincidentally matches.",
  "n_goals": "100 unique topics, 403 evidence conversations, 1,209 evidence-conversation segments, and 9,672 dialogue turns, with six categories of operation-level probes per instance",
  "tracking_demand": "The agent's memory system must track the trigger, target, scope, and state transition of each lifecycle operation (remember/forget/update/reflect) across up to 9,672 dialogue turns, maintaining a correct ordered memory-state trajectory rather than only a final answer.",
  "scoring": "subgoal-checkpoint-partial-credit -- six categories of operation-level probes evaluate each lifecycle-operation event individually (with gold operation traces), explicit sub-operation-level credit rather than a single final-answer score.",
  "horizon_value": "9,672 dialogue turns across 403 evidence conversations (100 unique topics), decomposed into 1,209 evidence-conversation segments",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "dataset-wide total (aggregated across conversations), not a single fixed per-conversation length",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce MemOps, a benchmark that reformulates conversational memory as a sequence of lifecycle operations and represents each memory event with a structured trace specifying its trigger, target, scope, state transition, and supporting evidence. A controllable generation pipeline embeds these operations into long, task-oriented conversations and produces gold operation traces together with six categories of operation-level probes.",
  "horizon_span": "The benchmark spans 100 unique topics and comprises 403 evidence conversations, which decompose into 1,209 evidence-conversation segments and 9,672 dialogue turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290178380",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285726166",
  "title": "MemoryArena: Benchmarking Agent Memory in Interdependent Multi-Session Agentic Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 62,
  "publication_date": "2026-02-18",
  "months_since_pub": 7,
  "citations_per_month": 8.86,
  "artifact_name": "MemoryArena",
  "artifact_kind": "environment/simulator",
  "domain": "personal-assistant-memory",
  "goal_types": "distill experience from earlier actions and feedback into memory during multi-session interaction; use previously distilled memory to guide later actions and solve subsequent, explicitly interdependent subtasks; solve overall tasks spanning web navigation, preference-constrained planning, progressive information search, and sequential formal reasoning",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Subtasks are explicitly interdependent: an agent must first acquire memory through earlier interaction/feedback in one session, then correctly recall and apply that memory to solve dependent subtasks in later sessions, so memory formation and later action are tightly coupled rather than independent stages.",
  "n_goals": null,
  "tracking_demand": "The agent must track which experiences it has distilled into memory across earlier sessions and correctly recall/apply the relevant portions of that memory to solve later, interdependent subtasks across multiple domains (web navigation, planning, search, formal reasoning).",
  "scoring": "other:task-success-across-multi-session-memory-agent-environment-loops",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Agents with near-saturated performance on existing long-context memory benchmarks like LoCoMo perform poorly in the MemoryArena agentic setting, though no single specific accuracy number or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "MemoryArena supports evaluation across web navigation, preference-constrained planning, progressive information search, and sequential formal reasoning, and reveals that agents with near-saturated performance on existing long-context memory benchmarks like LoCoMo perform poorly in our agentic setting, exposing a gap in current evaluations for agents with memory.",
  "horizon_span": "we introduce MemoryArena, a unified evaluation gym for benchmarking agent memory in multi-session Memory-Agent-Environment loops",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285726166",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288861940",
  "title": "Momento: Evaluating Persistent Memory and Reasoning with Multi-Session Agentic Conversations",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-05-30",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "Momento",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "take consequential, tool-mediated actions on behalf of a user within a multi-session service environment; resolve temporal dependencies between what happened in earlier sessions and what is being requested now; keep pace with evolving user goals across sessions rather than treating prior session history as static ground truth",
  "goal_origin": "mixed:given-up-front-service-tasks-with-user-goals-evolving-across-sessions",
  "decomposition": "sequential-chain",
  "interdependence": "Later-session actions depend on correctly re-validating (not just reusing) facts and decisions from earlier sessions, since the paper finds agents fail by 'treating prior session history as a reliable proxy for current context rather than stale information requiring re-validation'.",
  "n_goals": null,
  "tracking_demand": "Agent must track prior session history, recognize which parts of it may now be stale, and re-validate temporal dependencies and evolving user goals before taking consequential tool-mediated actions in the current session.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports a qualitative failure-mode finding (misestimation of user state) rather than a specific numeric or partial-credit scoring scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Current agents fail primarily through misestimation of user state \u2014 treating prior session history as reliable rather than re-validating it \u2014 revealing 'a substantial gap' versus realistic long-horizon human-agent interaction (no specific numeric score given).",
  "availability": null,
  "goal_span": "We introduce Momento, a benchmark for persistent agentic task completion in multi-session service environments, requiring agents to take consequential, tool-mediated actions while resolving temporal dependencies and evolving user goals across sessions.",
  "horizon_span": "existing benchmarks evaluate agents within a single session, ignoring past actions, stated preferences, and prior decisions that agents must integrate to fulfill personalized user goals",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288861940",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "284513236",
  "title": "MultiSessionCollab: Learning User Preferences with Memory to Improve Long-Term Collaboration",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 8,
  "publication_date": "2026-01-06",
  "months_since_pub": 8,
  "citations_per_month": 1.0,
  "artifact_name": "MultiSessionCollab",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "solve each of 20 sequential collaboration problems (one per session) for a given user; learn and apply that user's preferences across sessions to improve collaboration quality and reduce user effort over time",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each of a user's 20 sessions is a separate problem, but later sessions' collaboration quality depends on preferences learned and reflected upon in earlier sessions, so an agent's memory-update decisions in session N constrain how efficiently it collaborates in session N+1.",
  "n_goals": "20 randomly sampled problems per user (one per session, up to 10 conversational turns per session), totaling 10,000 collaborative sessions per evaluated agent",
  "tracking_demand": "The agent must track learned user preferences and reflections accumulated from prior sessions, applying them to reduce the number of conversational turns and user effort needed in each subsequent session across a 20-session sequence.",
  "scoring": "continuous-reward -- performance is measured via task success rate, interaction efficiency (turns needed), and reduced user effort over time (e.g. turns dropping from 10/8 to 6/4 by the third session with memory); the abstract does not describe a discrete subgoal-checkpoint rubric beyond these continuous efficiency/success metrics.",
  "horizon_value": "20 sessions per user (one problem per session, up to 10 conversational turns per session), totaling 10,000 collaborative sessions per agent across the benchmark; turns needed per session drop from 10/8 to 6/4 by the third session when memory is used",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per user (length of one user's whole 20-session collaboration sequence), each session up to 10 turns",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce MultiSessionCollab, a benchmark that evaluates how well agents can learn user preferences and leverage them to improve collaboration quality throughout multiple sessions... Extensive experiments show that equipping agents with our memory improves collaboration over time, yielding higher task success rates, more efficient interactions, and reduced user effort.",
  "horizon_span": "Each user collaborates with the agent to solve 20 randomly sampled problems, with one problem per session and a maximum of 10 conversational turns per session... this totals 10,000 collaborative sessions per agent.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284513236",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "sessions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290836900",
  "title": "PAST-Bench: Benchmarking the Foundations of Recursive Self-Improvement in Personal Agents",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 2,
  "publication_date": "2026-08-04",
  "months_since_pub": 1,
  "citations_per_month": 2.0,
  "artifact_name": "PAST-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "reuse a retained skill/procedure across sessions on a later fresh-session task; retrieve a previously stored preference/fact and apply it correctly in a new session; gather information in one session that is needed to complete a task in a later session; update outdated retained state (e.g. a stale fact) rather than acting on it unchanged",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later fresh-session tasks are designed to depend on experience retained (or deliberately withheld, in the matched 'off' condition) from earlier sessions, so whether a later task can be solved well depends on whether the agent correctly saved, retrieved, and updated state from prior sessions.",
  "n_goals": "26 scenarios and 204 episodes, across memory, procedural reuse, information gathering, and update capabilities",
  "tracking_demand": "The agent must save, retrieve, and update experience (preferences, task histories, tool routines, learned skills) across an ordered sequence of separate, fresh sessions, and the benchmark explicitly checks whether later-task gains actually follow this intended save/retrieve/update pathway rather than occurring by other means.",
  "scoring": "other:later-task-gain-with-pathway-evidence. The paper reports both later-task performance gains and whether those gains follow the intended save/retrieve/update pathway -- process-level as well as outcome-level credit, though not framed as a formal milestone rubric.",
  "horizon_value": "26 scenarios, 204 episodes total; ordered sequences of fresh-session tasks per scenario",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "across the whole benchmark (204 episodes spanning 26 scenarios); each scenario is itself an ordered multi-session sequence",
  "horizon_stated": "yes",
  "headline_result": "Hermes+ (Hermes plus five targeted interventions) raises the average gain from retained experience and gives clearer pathway evidence than base Hermes and other frameworks/models; no single percentage or human baseline given in the abstract.",
  "availability": "https://github.com/Gen-Verse/PAST-Bench",
  "goal_span": "Each agent runs through ordered sequences of fresh-session tasks under matched conditions that turn retained experience on and off. It spans 26 scenarios and 204 episodes across memory, procedural reuse, information gathering, and update. We report both later-task gains and whether those gains follow the intended save, retrieve, and update pathway.",
  "horizon_span": "It spans 26 scenarios and 204 episodes across memory, procedural reuse, information gathering, and update.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290836900",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "sessions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290690312",
  "title": "PAUSE: A User-Centric Benchmark for Personal AI Assistants in Unified Service Environments",
  "year": 2026,
  "venue": "Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-07-29",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "PAUSE",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "coordinate actions across heterogeneous user-owned services while respecting user-specific configurations and authorization/permission constraints; maintain consistency with evolving environment state across multi-turn interactions; for open-ended service-management tasks, satisfy semantic/behavioral trajectory-level goals; for constraint-intensive tasks, satisfy deterministic state-based verification conditions",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Actions across heterogeneous user-owned resources must remain consistent with shared environment state and authorization constraints, so an action taken on one resource/service can create or violate constraints relevant to subsequent actions on other resources within the same scenario.",
  "n_goals": "easy tasks average 2.11 dialogue rounds and 12.85 assistant tool calls; hard tasks average 5.23 dialogue rounds and 22.07 assistant tool calls",
  "tracking_demand": "The agent must track persistent user state, service-specific configurations and permissions, and prior actions taken across heterogeneous services, coordinating consistently across multi-turn interactions that scale from about 2 to over 5 dialogue rounds and roughly 13 to 22+ tool calls depending on difficulty.",
  "scoring": "other:multi-regime-evaluation -- open-ended service management tasks are scored via semantic and trajectory-level behavioral metrics while constraint-intensive tasks use deterministic, state-based verification, i.e. two distinct scoring regimes rather than one uniform binary rubric.",
  "horizon_value": "easy tasks average 2.11 dialogue rounds and 12.85 assistant tool calls; hard tasks average 5.23 dialogue rounds and 22.07 assistant tool calls",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "State-of-the-art proprietary models fail to reach 70% task completion on scenarios requiring stateful reasoning and configuration awareness; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "PAUSE captures core challenges of real-world assistant deployment by requiring agents to coordinate actions across heterogeneous user-owned resources while maintaining consistency with environment state and authorization constraints over multi-turn interactions... To support principled and reproducible evaluation, PAUSE adopts a multi-regime evaluation framework aligned with task characteristics. Open-ended service management tasks are assessed using semantic and trajectory-level behavioral metrics, while constraint-intensive tasks admit deterministic, state-based verification.",
  "horizon_span": "Avg. Rounds denotes the average number of dialogue turns ... consistently require more dialogue rounds and tool invocations across all evaluated models",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290690312",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290148518",
  "title": "PM-Bench: Evaluating Prospective Memory in LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-07-14",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "PM-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "maintain multiple ongoing and deferred intentions across a simulated week; execute a delayed intention at the correct future cue/state while continuing an ongoing activity; monitor latent environment changes relevant to deferred tasks",
  "goal_origin": "mixed:given-up-front-intentions-with-environment-emitted-cues",
  "decomposition": "set-of-independent",
  "interdependence": "The agent must continue an ongoing activity while simultaneously monitoring for cues that trigger any of several independently deferred tasks, so attention to the current activity competes with vigilance for deferred-task cues.",
  "n_goals": null,
  "tracking_demand": "Agent must track multiple deferred intentions and their trigger conditions, continuously monitor latent environment/state changes, and decide at each point whether any deferred task is now due, while continuing an ongoing activity across a simulated week.",
  "scoring": "other:f1-score",
  "horizon_value": "seven",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (one simulated week)",
  "horizon_stated": "yes",
  "headline_result": "Best method (a GPT-5.4 agent) reaches only 65.1% F1 across eight LLMs and eight agent configurations; no single strategy for improving prospective memory dominates across models; no human baseline given.",
  "availability": null,
  "goal_span": "Inspired by the Virtual Week paradigm from cognitive science, PM-Bench evaluates how well LLM agents maintain user intentions, execute delayed intentions, and monitor latent environment changes. Over the course of a simulated seven-day week, agents must continue an ongoing activity while deciding whether any deferred task is due.",
  "horizon_span": "Over the course of a simulated seven-day week, agents must continue an ongoing activity while deciding whether any deferred task is due.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290148518",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287023553",
  "title": "Proactive Agent Research Environment: Simulating Active Users to Evaluate Proactive Assistants",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 11,
  "publication_date": "2026-04-01",
  "months_since_pub": 5,
  "citations_per_month": 2.2,
  "artifact_name": "Pare-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "observe evolving app/user state via a stateful finite-state-machine simulation to infer the user's current goal; correctly time an intervention (neither too early nor too late) once a need is inferred; orchestrate actions across multiple apps (communication, productivity, scheduling, lifestyle) to address the inferred goal",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "hierarchical",
  "interdependence": "The user simulator's state-dependent action space means later user actions and needs depend on the app's evolving state (and on the agent's own prior interventions), so goal-inference and intervention-timing decisions must track this evolving, stateful context across turns rather than treating each observation independently.",
  "n_goals": "143 diverse tasks spanning communication, productivity, scheduling, and lifestyle apps",
  "tracking_demand": "The agent must continuously observe the simulated user's stateful, sequential app interactions to infer an emerging goal, decide the right moment to intervene, and coordinate the intervention across multiple apps.",
  "scoring": "other:context-observation-and-intervention-timing-metrics. The abstract states the benchmark is 'designed to test context observation, goal inference, intervention timing, and multi-app orchestration,' multiple distinct evaluated capabilities, but does not describe a single subgoal-checkpoint credit scheme.",
  "horizon_value": "simulation runs for a maximum of 10 turns",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (per task simulation)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "Pare models applications as finite state machines with stateful navigation and state-dependent action space for the user simulator, enabling active user simulation. Building on this foundation, we present Pare-Bench, a benchmark of 143 diverse tasks spanning communication, productivity, scheduling, and lifestyle apps, designed to test context observation, goal inference, intervention timing, and multi-app orchestration.",
  "horizon_span": "The simulation runs for a maximum of 10 turns.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287023553",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291324855",
  "title": "PersonaMem-v3: Toward Omni-Platform Personal Intelligence for Holistic User Understanding, Recommendation, and Agentic Tasks",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-07-16",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "PersonaMem-v3",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "build holistic cross-platform user understanding from social media, chatbot, calendar, and AI-companion engagement histories; personalize responses to reflect the user's evolving preferences over time; rerank recommendations on social media in a steerable way; act proactively across platforms when appropriate; hold back from personalizing when it would be inappropriate, repetitive, outdated, or unnecessary",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "All evaluated capabilities (personalization, recommendation, proactiveness, restraint) are grounded in the same time-indexed, cross-platform user history, so correctly personalizing or holding back at a later time point depends on correctly having tracked how the user's preferences evolved across earlier platforms and time.",
  "n_goals": null,
  "tracking_demand": "The agent must track a time-indexed model of the user's preferences, intents, habits, and social relationships as they evolve across multiple platforms (social media, chatbot, calendar, AI-companion), and decide when NOT to act or personalize.",
  "scoring": "other:multi-capability-evaluation(personalization/recommendation/proactiveness/restraint/geo-temporal). The benchmark brings personalization, recommendation, proactiveness, agentic tool use, and geo-temporal reasoning into one framework anchored in psychology/social-linguistics/user-behavior theories -- multiple distinct evaluated capabilities rather than one binary score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "The benchmark brings personalization, LLM-powered recommendation, proactiveness, agentic tool use, and geo-temporal reasoning into one framework, anchored in psychology, social-linguistics, and user-behavior theories. It evaluates whether AI agents can infer holistic user understanding from cross-platform evidence, personalize responses, rerank recommendations on social media, follow user steering through natural language, and hold back when personalization would be inappropriate, repetitive, outdated, or unnecessary.",
  "horizon_span": "PersonaMem-v3 is seeded from more than one million anonymized real-world engagement histories, most of which are implicit signals, and uses them to construct time-indexed user digital worlds",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291324855",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285285673",
  "title": "ProAgentBench: Evaluating LLM Agents for Proactive Assistance with Real-World Data",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 10,
  "publication_date": "2026-02-04",
  "months_since_pub": 7,
  "citations_per_month": 1.43,
  "artifact_name": "ProAgentBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "predict the correct timing for a proactive intervention within a continuous workflow; generate appropriate assist content once an intervention point is identified",
  "goal_origin": "implied-by-constraints",
  "decomposition": "hierarchical",
  "interdependence": "The hierarchical task framework requires timing prediction to happen correctly before assist-content generation is even attempted, and both must be conditioned on the pre-assistance behavioral context accumulated over the continuous (bursty) real-user workflow rather than an isolated task instance.",
  "n_goals": "2 (timing prediction; assist-content generation), evaluated over 28,000+ events",
  "tracking_demand": "Agent must model long-term memory and historical, pre-assistance behavioral context (bursty interaction patterns, B=0.787) to decide both whether/when to intervene and what to say.",
  "scoring": "other:not-stated \u2014 the abstract reports that long-term memory/historical context 'significantly enhance prediction accuracy' but does not describe a specific partial-credit or checkpoint scheme for the two decomposed sub-tasks.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Long-term memory and historical context significantly enhance prediction accuracy; real-world training data substantially outperforms synthetic alternatives (no single numeric headline score given).",
  "availability": "https://anonymous.4open.science/r/ProAgentBench-6BC0",
  "goal_span": "a hierarchical task framework that decomposes proactive assistance into timing prediction and assist content generation",
  "horizon_span": "a privacy-compliant dataset with 28,000+ events from 500+ hours of real user sessions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285285673",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "290377727",
  "title": "ProEvent: An Event-centric Benchmark for Proactive Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 2,
  "publication_date": "2026-07-20",
  "months_since_pub": 2,
  "citations_per_month": 1.0,
  "artifact_name": "ProEvent",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "identify new upcoming events, including implicit ones, from ongoing instant-messaging chats; maintain and update a timetable of a user's events over time; time proactive responses correctly (neither too early nor too late); handle event cancellations correctly rather than overacting; produce correct single-step and multi-step responses per event",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Multiple concurrent chat threads and events share the same user timetable, so correctly updating or cancelling one event requires distinguishing it from noise and from other concurrent threads without cross-contaminating the timetable.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain a live, updatable timetable of a user's upcoming events, tracking concurrent chat threads, noise, and event cancellations, and decide both when and how (single- vs. multi-step) to respond as messages arrive.",
  "scoring": "other:response-timing-plus-single-and-multi-step-response-correctness. The benchmark evaluates on three separate axes (response timing, single-step response correctness, multi-step response correctness), a multi-dimensional non-binary rubric rather than one aggregate score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Even GPT-5.1 reacts correctly in only 26.7% of scenarios; current agents frequently overact and struggle with event cancellation; no human baseline given.",
  "availability": null,
  "goal_span": "ProEvent provides synthesized yet realistic chats that consider the dynamic interaction among users, concurrent chat threads, and noise in the real world, and evaluates proactive agents on response timing, single-step response correctness, and multi-step response correctness.",
  "horizon_span": "synthesized yet realistic chats that consider the dynamic interaction among users, concurrent chat threads, and noise in the real world",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290377727",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284647968",
  "title": "RealMem: Benchmarking LLMs in Real-World Memory-Driven Interaction",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 16,
  "publication_date": "2026-01-11",
  "months_since_pub": 8,
  "citations_per_month": 2.0,
  "artifact_name": "RealMem",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "track evolving project goals across long-term, cross-session dialogues; manage dynamic context dependencies (schedule/memory) inherent to real-world projects; respond correctly to natural user queries grounded in accumulated project history across eleven scenarios",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later cross-session dialogue turns depend on project state, schedule, and memory established in earlier sessions, so failing to track evolving project goals or schedule dependencies causes inconsistent answers to later natural queries.",
  "n_goals": "over 2,000 cross-session dialogues across eleven scenarios",
  "tracking_demand": "The system must track long-term project states and dynamic context/schedule dependencies across more than 2,000 cross-session dialogues, since project goals evolve over time rather than remaining fixed.",
  "scoring": "other:not-specified-in-abstract",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/AvatarMemory/RealMemBench",
  "goal_span": "RealMem comprises over 2,000 cross-session dialogues across eleven scenarios, utilizing natural user queries for evaluation.",
  "horizon_span": "RealMem comprises over 2,000 cross-session dialogues across eleven scenarios",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284647968",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "290648393",
  "title": "Setoka: A Benchmark for Hierarchical User Understanding in Personalized Agents over Heterogeneous Data",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-07-29",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "Setoka",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "retrieve explicit facts from past interactions (semantic memory); recall specific past episodes/events accurately (episodic memory); infer recurring behavior patterns from heterogeneous data over time (behavior pattern); infer abstract personality traits from heterogeneous, fragmented information (personality trait)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Higher-level understanding (behavior pattern, personality trait) requires integrating heterogeneous and fragmented information dispersed over time and across the lower memory levels (semantic facts, episodic events), so failure to correctly retrieve/recall lower-level memory undermines the higher-level abstraction tasks built on it.",
  "n_goals": "four levels of user understanding (semantic memory, episodic memory, behavior pattern, personality trait); 3 language models x 5 memory systems x 10 synthetic users evaluated",
  "tracking_demand": "The memory-augmented agent must retain and integrate heterogeneous user data (explicit facts, episodic events, behavioral observations) dispersed over long-term interaction history to answer queries at each of the four hierarchical understanding levels.",
  "scoring": "other:per-level-accuracy-across-four-hierarchical-understanding-levels",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Existing systems perform well on semantic memory retrieval but decline on episodic memory, and decline further on behavior-pattern and personality-trait understanding tasks; no single specific top score or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "Setoka defines four levels of user understanding, i.e., semantic memory, episodic memory, behavior pattern, and personality trait... Our comprehensive evaluation reveals that while existing systems perform well on semantic memory retrieval, their performance declines on episodic memory.",
  "horizon_span": "motivating the design of memory mechanisms for cross-source integration and abstraction over long-term user behavior",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290648393",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "286572981",
  "title": "Shopping Companion: Benchmarking and Training LLM Agents for Long-Horizon Preference-Grounded E-Commerce Tasks",
  "year": 2026,
  "venue": null,
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 4,
  "publication_date": "2026-03-16",
  "months_since_pub": 6,
  "citations_per_month": 0.67,
  "artifact_name": "Shopping Companion Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "recommend products correctly aligned with preferences expressed across long-horizon conversations; manage a budget while shopping; assemble bundle deals satisfying multiple items' constraints jointly; correctly recall and apply user preferences carried over from earlier sessions (cross-session preference memory)",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Later-session shopping decisions (recommendations, budget management, bundle deals) depend on correctly recalled preferences from earlier sessions, and the paper's two identified failure modes (preference hallucination causing cascading errors, insufficient verification of product attributes) show that an error in preference recall propagates into subsequent tool-call decisions.",
  "n_goals": "2 shopping tasks requiring cross-session preference memory, over a product pool of 1.2 million+ real-world items",
  "tracking_demand": "Agent must accumulate and correctly recall user shopping preferences across sessions, and verify product attributes against user requirements at each tool call, to avoid cascading preference-hallucination errors.",
  "scoring": "other:mixed \u2014 'annotation-free, tool-wise rewards that provide process supervision for each tool call', i.e. explicit per-tool-call (subgoal-level) process credit designed specifically to address 'reward sparsity in long-horizon tasks', alongside an overall task success rate.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even state-of-the-art models such as GPT-5 achieve success rates below 70%; a fine-tuned lightweight 4B model consistently outperforms strong baselines in both preference capture and task performance (no human baseline given).",
  "availability": null,
  "goal_span": "we design annotation-free, tool-wise rewards that provide process supervision for each tool call, alleviating reward sparsity in long-horizon tasks",
  "horizon_span": "the absence of benchmarks for evaluating long-term preference-aware shopping tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286572981",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289923148",
  "title": "SovereignNegotiation-Bench: Evaluating User-Owned Personal Agents In Delegated Bargaining Under Privacy, Consent, Evidence, And Institutional Pressure",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-07-02",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "SovereignNegotiation-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "reach a negotiated agreement on the user's behalf (e.g., cost splits, refunds, subscription changes); preserve user utility while negotiating; avoid privacy leakage and consent violations during negotiation; ground claims in evidence and maintain auditability of the negotiation trace; escalate appropriately rather than over-concede under institutional pressure",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Agreement success, user utility, privacy, consent, evidence grounding, concession discipline, escalation, and auditability are all jointly evaluated over the same multi-turn negotiation trace, so maximizing agreement (e.g., by over-conceding or leaking private information) can directly harm the other tracked objectives.",
  "n_goals": "8 jointly tracked evaluation objectives (agreement, user utility, privacy, consent, evidence grounding, concession discipline, escalation, auditability); 240 scenarios, 4 model families, 14 baselines",
  "tracking_demand": "The agent must track its private utilities/disclosure constraints, evidence requirements, and institutional-pressure cues across a multi-turn negotiation trace, keeping agent-visible observable state separate from evaluator-only labels.",
  "scoring": "other:multi-dimensional-sovereign-negotiation-score. The benchmark evaluates agreement success jointly with several other axes (user utility, privacy, consent, evidence grounding, concession discipline, escalation, auditability) via a blinded 3-annotator audit -- an explicit multi-axis, non-binary rubric rather than agreement success alone.",
  "horizon_value": "61,135 parsed action rows across 13,440 frozen-prompt live trajectories (~4.5 actions/trajectory)",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per whole episode (per negotiation trajectory); derived by dividing total parsed action rows by total trajectories, since the paper reports only the totals directly.",
  "horizon_stated": "yes",
  "headline_result": "The strongest agreement-maximizing baseline achieves the highest agreement rate but low user utility and high privacy/consent risk; FullSovereign does not maximize agreement but obtains the best sovereign negotiation score by preserving utility, minimizing leakage, grounding claims, and reducing unauthorized commitments; no external human baseline given.",
  "availability": null,
  "goal_span": "The benchmark separates agent-visible observable state from evaluator-only labels and evaluates agreement success jointly with user utility, privacy, consent, evidence grounding, concession discipline, escalation, and auditability.",
  "horizon_span": "We report an artifact-backed validation over 240 scenarios, 4 model families, 14 baselines, 13,440 frozen-prompt live trajectories, 61,135 parsed action rows, and a blinded 3-annotator audit over 300 items.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289923148",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289933130",
  "title": "SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-07-06",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "SovereignPA-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "advance a user's current, evolving interests while respecting privacy boundaries, consent constraints, and evidence requirements; minimize user burden while resisting manipulative platform incentives; preserve auditability of decisions across 120 sovereignty stress scenarios",
  "goal_origin": "mixed:given-up-front-user-goal-plus-environment-emitted-platform-mediation-events",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Consent and privacy boundaries constrain what evidence-gathering or platform-mediated actions are permissible, and burden/manipulation-resistance goals trade off against advancing the user's evolving interest, so the eight component metrics (task success, alignment, privacy, consent, evidence, manipulation, burden, auditability) are jointly, not independently, satisfied within one scenario.",
  "n_goals": "120 sovereignty stress scenarios across 4 model families and 8 policy baselines, yielding 3,840 frozen-prompt trajectories",
  "tracking_demand": "Agent must track the user's evolving intent, what has been disclosed to which platform/party (ObservableState vs. evaluator-only HiddenLabels), consent already given, and accumulated burden across a scenario, all while resisting manipulative incentives.",
  "scoring": "other:eight-component-metric-scoring - reports component metrics for task success, alignment, privacy, consent, evidence, manipulation, burden, and auditability separately, plus a blinded 3-annotator audit over 240 items; an explicit multi-dimensional (partial-credit-like) scoring scheme rather than one binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Full-sovereign scaffolding improves sovereignty score over direct, memory-only, consent-only, evidence-only, ReAct/tool-use, safety-prompt, and judge-guard baselines while reducing privacy leakage, consent violation, over-concession, and manipulation capture; specific numeric scores are not given in the abstract; human audit shows high agreement on privacy/consent and lower agreement on manipulation.",
  "availability": null,
  "goal_span": "The benchmark separates agent-visible ObservableState from evaluator-only HiddenLabels, reports component metrics for task success, alignment, privacy, consent, evidence, manipulation, burden, and auditability, and preserves paired scenario ordering for model and policy comparisons.",
  "horizon_span": "We evaluate 120 sovereignty stress scenarios across 4 model families and 8 policy baselines, yielding 3,840 frozen-prompt trajectories with raw prompts, outputs, provider-form responses, parsed actions, recomputable metrics, hard-set analyses, qualitative cases, and a blinded 3-annotator audit over 240 items.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289933130",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291257456",
  "title": "Can Agent Memory Systems Track Evolving State?",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-08-20",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "StateMemBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "track the evolving state of facts, constraints, and decisions as they are revised over a long multi-session interaction; answer questions reflecting the CURRENT state, not a superseded prior state",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Facts/constraints/decisions are revised over time (supersession), so answering correctly at a later session depends on correctly tracking which earlier statement has since been overwritten and by what.",
  "n_goals": "234 multi-session scenarios spanning two conversation-length regimes",
  "tracking_demand": "Agent's memory system must track the current value of each evolving fact/constraint/decision plus its supersession and relational dependencies, distinguishing current from superseded state at each query point.",
  "scoring": "other:closed-pool-current-vs-superseded-vs-fail-grading - closed-pool grading explicitly scores whether an answer reflects the current state, the superseded state, or fails otherwise, separating state-tracking failures from other errors by construction; this is a structured 3-way per-item grading scheme rather than simple binary success.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "StateMem improves current-state accuracy over the strongest same-backbone baseline by 1.8x (0.205 -> 0.363) on DeepSeek-V4-Flash and over the strongest memory system by 1.6x (0.149 -> 0.233) on Qwen-3.5-9B; as a lightweight wrapper it lifts current-state accuracy by +32 to +67 points across six memory/retrieval backends, with +15 to +32 of those points attributed to state structure by a length/cost-matched control.",
  "availability": null,
  "goal_span": "we argue an effective memory system must track the evolving state of the world; as facts, constraints, and decisions are revised over a long interaction, answers must reflect the current state and not a superseded one. We define this capability as state tracking and instantiate it in StateMemBench, a benchmark of 234 multi-session scenarios spanning two conversation-length regimes.",
  "horizon_span": "We define this capability as state tracking and instantiate it in StateMemBench, a benchmark of 234 multi-session scenarios spanning two conversation-length regimes.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291257456",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289263870",
  "title": "StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-06-12",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "StreamMemBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "correctly recall/use evidence observed in an initial task drawn from a streaming egocentric anchor; incorporate feedback/interaction experience from the initial task into a later follow-up task; carry evidence forward from what the agent observes and how the user interacts with it, to future similar tasks",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The follow-up task's success is explicitly conditioned on whether feedback and interaction experience from the initial task were correctly reused, so the second goal is directly dependent on the outcome/state of the first; the paper reports systems often fail to convert even correctly-stored initial evidence into reliable follow-up behavior.",
  "n_goals": "two-step task sequence (initial + follow-up) per evidence anchor from EgoLife egocentric streams",
  "tracking_demand": "The agent must carry stored evidence and interaction feedback forward from an initial task to a corresponding follow-up task drawn from continuous streaming egocentric observations, diagnosed via four metrics (evidence recall, initial evidence use, feedback incorporation, follow-up reuse).",
  "scoring": "other:four-metric-diagnostic-scoring",
  "horizon_value": "2 (initial task + follow-up task) per evidence anchor",
  "horizon_unit": "other:task-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per evidence anchor (a short two-step task pair drawn from a longer continuous egocentric stream)",
  "horizon_stated": "yes",
  "headline_result": "Current systems often fail to use observed evidence or turn feedback into reliable follow-up behavior, even when evidence is stored or feedback is incorporated locally; no single numeric headline figure given.",
  "availability": "https://github.com/landian60/StreamMemBench",
  "goal_span": "We introduce StreamMemBench, a streaming benchmark that constructs a two-step task sequence around each evidence anchor from EgoLife egocentric streams. The initial task tests evidence use, while the follow-up task tests whether feedback and interaction experience are reused. Four metrics diagnose evidence recall, initial evidence use, feedback incorporation, and follow-up reuse.",
  "horizon_span": "StreamMemBench, a streaming benchmark that constructs a two-step task sequence around each evidence anchor from EgoLife egocentric streams. The initial task tests evidence use, while the follow-up task tests whether feedback and interaction experience are reused.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": "This benchmark's per-anchor unit is only a two-step (initial + follow-up) task sequence, which is a notably short horizon relative to the corpus's long-horizon, multi-goal focus, even though the underlying EgoLife streaming source itself spans much longer continuous observation; flagging for the judge's attention.",
  "url": "https://api.semanticscholar.org/CorpusId:289263870",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289668750",
  "title": "Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 7,
  "publication_date": "2026-06-25",
  "months_since_pub": 3,
  "citations_per_month": 2.33,
  "artifact_name": "Supersede",
  "artifact_kind": "environment/simulator",
  "domain": "personal-assistant-memory",
  "goal_types": "answer using the current (most up-to-date) value of a fact that changes over time (e.g., a user's address, a price, a plan); discard/avoid using superseded (stale) fact values; maintain a bounded, self-maintained memory that keeps pace as the conversation grows",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Each new update to a fact supersedes its prior value, so correctness requires recognizing which of possibly many prior mentions of the same fact is authoritative; sheer conversation growth (24x) compounds the risk of using stale values even when overall comprehension is not the bottleneck.",
  "n_goals": null,
  "tracking_demand": "The agent must maintain a bounded, self-maintained memory of facts across long, multi-session interactions and always resolve to a fact's most current value, discarding superseded ones, as the conversation grows arbitrarily long.",
  "scoring": "other:accuracy-on-knowledge-update-subset-of-LongMemEval-plus-RL-reward-for-current-value-answers",
  "horizon_value": "conversation length grows 24x (accuracy falls from 68% to 28% over this range, n=25)",
  "horizon_unit": "other:relative-conversation-length-growth-factor",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode (relative growth across a single conversation being extended)",
  "horizon_stated": "yes",
  "headline_result": "On LongMemEval's knowledge-update subset, replacing full context with bounded self-maintained memory drops accuracy from 92% to 77% even for a frontier model (gpt-5.4); GRPO fine-tuning a small open model (Qwen2.5-3B) nearly doubles held-out supersession accuracy (9.0% to 16.7%). No human/expert baseline is given.",
  "availability": null,
  "goal_span": "We release Supersede, an open reinforcement-learning environment (on the verifiers / prime-rl stack) that turns this measurement into a training signal: agents are rewarded for answering from the current value and penalized for stale ones.",
  "horizon_span": "as the conversation grows 24x, accuracy falls further (from 68% to 28%), and granting the agent proportionally more memory yields no detectable recovery",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289668750",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291167182",
  "title": "When Personal Memory Has No Single Answer: Evaluating LLM Agents under Irreducible Conflict",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-08-14",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "TANGLE",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "recognize underdetermination when personal memory has no single answer; retain/preserve conflicting alternatives rather than collapsing to one definitive answer; seek clarification rather than acting on unjustified overconfidence; choose an action appropriate to Context-Partitioned, Behavior-Oscillation, or Source-Contradiction conflict types",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "In the pipeline track, memory must first be correctly extracted from multi-session dialogues before conflict perception, causal reasoning, and action choice can be evaluated, so extraction fidelity constrains all downstream reasoning dimensions.",
  "n_goals": "541 instances across 40 personas and three conflict types (CPC, BOC, SCC), evaluated across five dimensions",
  "tracking_demand": "Agent must preserve rather than resolve conflicting evidence, monitor five behavior dimensions (conflict perception, causal reasoning, confidence calibration, clarification seeking, memory faithfulness), and in the pipeline track extract and preserve conflict-bearing relations from multi-session dialogues.",
  "scoring": "other:five-dimension-rubric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we evaluate two tracks---an oracle track with curated memory and a pipeline track that extracts memory from multi-session dialogues---on five dimensions: conflict perception, causal reasoning, confidence calibration, clarification seeking, and memory faithfulness.",
  "horizon_span": "a pipeline track that extracts memory from multi-session dialogues",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291167182",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286775607",
  "title": "VehicleMemBench: An Executable Benchmark for Multi-User Long-Term Memory in In-Vehicle Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-03-25",
  "months_since_pub": 6,
  "citations_per_month": 0.17,
  "artifact_name": "VehicleMemBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "model multi-user preferences continuously as they evolve over time; resolve inter-user preference conflicts; correctly invoke 23 tool modules to reach a predefined target environment state; track changing user habits across many historical memory events",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Multiple users' preferences can directly conflict, and correct actions must reconcile with preferences/habits established or changed earlier across over 80 historical memory events per sample.",
  "n_goals": "23 tool modules; over 80 historical memory events per sample",
  "tracking_demand": "Agent must track evolving, sometimes conflicting, per-user preferences and habits across 80+ historical memory events, verifying its actions by comparing the resulting environment state to a predefined target state.",
  "scoring": "binary-final-success",
  "horizon_value": "over 80",
  "horizon_unit": "other:historical-memory-events-per-sample",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per episode/sample (per benchmark instance)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "The benchmark evaluates tool use and memory by comparing the post-action environment state with a predefined target state, enabling objective and reproducible evaluation without LLM-based or human scoring. VehicleMemBench includes 23 tool modules, and each sample contains over 80 historical memory events.",
  "horizon_span": "each sample contains over 80 historical memory events",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286775607",
  "provenance": "forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291024322",
  "title": "VibeLifeBench: Can Your Life Agent Be Proactive and Persistent in a Living World?",
  "year": 2026,
  "venue": "",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-08-11",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "VibeLifeBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "complete 200 long-horizon tasks across ten everyday-life domains over scripted multi-week timelines; proactively decide when to act, ask, or stay silent without being explicitly prompted; notice unannounced/silent world changes by re-inspecting the world; keep one plan coherent from the first day to the last while upholding unstated implicit constraints",
  "goal_origin": "implied-by-constraints",
  "decomposition": "open-ended",
  "interdependence": "The simulated world advances on its own clock and many changes are silent, so acting correctly later depends on the agent having proactively noticed earlier unannounced changes; failing to re-inspect breaks the coherence of the single plan spanning the whole task.",
  "n_goals": "200 tasks across ten everyday-life domains",
  "tracking_demand": "Agent must track end-state goals, the timeliness of its own actions, and implicit constraints across scripted multi-week timelines in a simulated world of 22 mock services that changes on its own clock, much of it silently.",
  "scoring": "other:fine-grained-weighted-checks",
  "horizon_value": "multi-week (200 scripted tasks)",
  "horizon_unit": "other:multi-week-scripted-timeline",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (each task is a scripted multi-week timeline)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "Every task is graded by fine-grained, weighted checks that read only what the agent actually left behind, covering the end state, the timeliness of its actions, and whether it upheld the implicit constraints.",
  "horizon_span": "We introduce VibeLifeBench, a benchmark of 200 long-horizon tasks across ten everyday-life domains. Each task is a scripted multi-week timeline in a simulated world of 22 mock services.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291024322",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288672451",
  "title": "VitaBench 2.0: Evaluating Personalized and Proactive Agents in Long-Term User Interactions",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 5,
  "publication_date": "2026-05-26",
  "months_since_pub": 4,
  "citations_per_month": 1.25,
  "artifact_name": "VitaBench 2.0",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "continuously extract, utilize, and update evolving user preferences across a temporally ordered sequence of tasks for one user; proactively recognize missing information and actively acquire it from users/environment before making a decision",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each task in a user's sequence can add, delete, or modify preferences that constrain how later tasks in the same sequence should be completed, so correctly serving a later task depends on having correctly extracted and updated preference state from earlier tasks.",
  "n_goals": "56 users, 819 subtasks total, 66 tools, 2,000+ fine-grained preferences; each user has at least 10 tasks in their temporally ordered task sequence",
  "tracking_demand": "The agent must continuously extract, update, and apply an evolving set of user preferences (which can be added, deleted, or modified between tasks) across a temporally ordered sequence of at least 10 tasks per user, and proactively recognize when it needs to ask for missing information.",
  "scoring": "other:extensible-memory-interface-comparison -- the paper provides an extensible memory interface enabling controlled comparison across different memory architectures, implying architecture-level comparative scoring rather than a single stated subgoal-checkpoint rubric within one task sequence.",
  "horizon_value": "users have at least 10 tasks in their temporally ordered task sequences (56 users, 819 subtasks total)",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per user (length of one user's whole temporally ordered task sequence)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/meituan-longcat/VitaBench-2.0",
  "goal_span": "In VitaBench 2.0, tasks are organized as temporally ordered sequences for individual users, where preferences are embedded in fragmented and heterogeneous interactions. Successful completion of tasks requires the agent to continuously extract, utilize, and update user preferences from these interactions. We further evaluate proactiveness through tasks that require agents to recognize missing information and actively acquire it from users or environments before making decisions.",
  "horizon_span": "We report the task index of up to 10, as users have at least 10 tasks in their task sequences.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288672451",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291602169",
  "title": "WorldBench: Culturally Grounded Benchmark for Multilingual Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 0,
  "publication_date": "2026-09-01",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "WorldBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "complete genuine, persona-grounded everyday workflows across seven languages and eight cultures; preserve sandbox environment state while acting via structured actions; minimize unnecessary modification/side effects while completing the requested task",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Constrained Task Success combines task completion with state preservation, so an action satisfying the immediate instruction can still fail overall if it introduces unwanted side effects/modifications elsewhere in the sandboxed environment.",
  "n_goals": "1,600 tasks across seven languages and eight cultures",
  "tracking_demand": "Agent must track sandbox state to complete persona-grounded workflows correctly while minimizing unwanted modifications, across long-horizon tasks and language/culture variation.",
  "scoring": "other:constrained-task-success-metric",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier models reach only 49.2% CTS, with all models showing large gaps between correctness and environment preservation; no human baseline given.",
  "availability": null,
  "goal_span": "we introduce Constrained Task Success (CTS), which combines natural language instructions and testbeds to score task completion, minimal modification, and other complementary metrics through deterministic and LLM-as-a-Judge evaluations.",
  "horizon_span": "current agents remain brittle in multilingual, agentic scenarios, especially for long-horizon tasks and under state-preservation constraints",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291602169",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288741642",
  "title": "WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 1,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "WorldMemArena",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "track and update an evolving personal/task state across a lifelong-evolution scenario; write, maintain, retrieve, and use memory correctly through the four-stage Action-World Interaction Loop; use visual evidence from real observations, actions, and feedback (Agentic Execution); answer QA correctly against gold memory points while resisting annotated distractors",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The four-stage lifecycle (write, maintain, retrieve, use) is explicitly ordered: a maintenance failure (failing to update stale memory) corrupts what can later be retrieved and used, and the paper's stage-level diagnosis exists specifically to localize failures to one of these four dependent stages.",
  "n_goals": "400 multi-session multimodal tasks",
  "tracking_demand": "The agent must write new memory from observations/actions/feedback, maintain it against evolving personal/task state (Lifelong Evolution), and correctly retrieve/use it later, annotated against gold memory points, updates, distractors, and evidence chains across multiple sessions.",
  "scoring": "other:stage-level-diagnosis-write-maintain-retrieve-use",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Better memory writing/storage does not guarantee better performance; multimodal memory still struggles to fully use visual evidence; harness memory is more flexible but costly and less reliable; no single numeric headline given.",
  "availability": "https://github.com/UCSB-AI/WorldMemArena",
  "goal_span": "we formulate multimodal agent memory as an Action-World Interaction Loop with an observable four-stage lifecycle, and instantiate it in WorldMemArena: 400 multi-session multimodal tasks spanning Lifelong Evolution (evolving personal and task states) and Agentic Execution (memory from real observations, actions, and feedback), annotated with gold memory points, updates, distractors, and evidence chains for stage-level diagnosis.",
  "horizon_span": "instantiate it in WorldMemArena: 400 multi-session multimodal tasks spanning Lifelong Evolution ... and Agentic Execution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288741642",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288254271",
  "title": "\u03c0-Bench: Evaluating Proactive Personal Assistant Agents in Long-Horizon Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 3,
  "publication_date": "2026-05-14",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "\u03c0-Bench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "identify and act on hidden/unstated user needs before they are explicitly stated; complete each of 100 multi-turn tasks across 5 domain-specific user personas; resolve inter-task dependencies across sessions; maintain continuity of prior interaction context across sessions to resolve later proactive intents",
  "goal_origin": "mixed:given-up-front-with-hidden-intents-implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks have explicit inter-task dependencies and cross-session continuity, so correctly resolving a hidden intent in an earlier task/session affects whether the agent can proactively address dependent needs in later tasks/sessions.",
  "n_goals": "100 multi-turn tasks across 5 domain-specific user personas",
  "tracking_demand": "The agent must track hidden/unstated user intents, inter-task dependencies, and information carried over across sessions, jointly measuring proactivity (anticipating needs) and task completion (executing them) over extended interactions.",
  "scoring": "other:joint-proactivity-and-task-completion-metrics",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "By incorporating hidden user intents, inter-task dependencies, and cross-session continuity, $\\pi$-Bench evaluates agents'ability to anticipate and address user needs over extended interactions, jointly measuring proactivity and task completion in long-horizon trajectories that better reflect real-world use.",
  "horizon_span": "$\\pi$-Bench, a benchmark for proactive assistance comprising 100 multi-turn tasks across 5 domain-specific user personas",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288254271",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "283261706",
  "title": "Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 124,
  "publication_date": "2025-11-25",
  "months_since_pub": 10,
  "citations_per_month": 12.4,
  "artifact_name": "Evo-Memory",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "search, adapt, and evolve memory after each interaction in a sequential task stream; solve each task drawn from 10 diverse multi-turn goal-oriented and single-turn reasoning/QA datasets; reuse experience accumulated from earlier tasks to improve on later tasks in the same stream",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later tasks in a stream can benefit from (or be hindered by) memory accumulated from earlier tasks; both the ExpRAG baseline and the proposed ReMem method depend on correctly retrieving and reusing prior experience for subsequent task performance.",
  "n_goals": "10 diverse multi-turn goal-oriented and single-turn reasoning/QA datasets; over 10 representative memory modules compared",
  "tracking_demand": "Agent must search, adapt, and evolve its own memory continuously after each interaction across a sequential task stream, integrating reasoning, task actions, and memory updates to achieve continual improvement.",
  "scoring": "other:task-accuracy-across-streamed-datasets",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Evo-Memory structures datasets into sequential task streams, requiring LLMs to search, adapt, and evolve memory after each interaction... we unify and implement over ten representative memory modules and evaluate them across 10 diverse multi-turn goal-oriented and single-turn reasoning and QA datasets.",
  "horizon_span": "LLMs are required to handle continuous task streams, yet often fail to learn from accumulated interactions, losing valuable contextual insights, a limitation that calls for test-time evolution",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283261706",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "283897267",
  "title": "Forgetful but Faithful: A Cognitive Memory Architecture and Benchmark for Privacy-Aware Generative Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 11,
  "publication_date": "2025-12-14",
  "months_since_pub": 9,
  "citations_per_month": 1.22,
  "artifact_name": "Forgetful but Faithful Agent (FiFA) benchmark",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "maintain narrative coherence across long-term interaction; complete multi-step goals despite a bounded memory budget; preserve social recall accuracy under six candidate forgetting policies; preserve privacy by not retaining/leaking information beyond its retention schema; minimize cost while balancing the above under different memory budgets",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Memory-budget constraints create a shared resource across all five evaluated dimensions: retaining more for narrative coherence or goal completion competes with cost efficiency and privacy preservation, so forgetting-policy choices trade off across the whole goal set.",
  "n_goals": "300 evaluation runs across multiple memory budgets and agent configurations",
  "tracking_demand": "The agent must track its own memory budget consumption, which information to retain vs. forget under its chosen policy, ongoing multi-step goal progress, and social/privacy-relevant facts, across long-term interactive scenarios.",
  "scoring": "continuous-reward",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "The hybrid forgetting policy achieves a composite score of 0.911, the best among six candidate policies; no explicit human baseline given.",
  "availability": null,
  "goal_span": "We present the Forgetful but Faithful Agent (FiFA) benchmark, a comprehensive evaluation framework that assesses agent performance across narrative coherence, goal completion, social recall accuracy, privacy preservation, and cost efficiency. Through extensive experimentation involving 300 evaluation runs across multiple memory budgets and agent configurations, we demonstrate that our hybrid forgetting policy achieves superior performance (composite score: 0.911)",
  "horizon_span": "Through extensive experimentation involving 300 evaluation runs across multiple memory budgets and agent configurations",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283897267",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281682069",
  "title": "Mem-\u03b1: Learning Memory Construction via Reinforcement Learning",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 27,
  "publication_date": "2025-09-30",
  "months_since_pub": 12,
  "citations_per_month": 2.25,
  "artifact_name": "Mem-alpha",
  "artifact_kind": "dataset",
  "domain": "personal-assistant-memory",
  "goal_types": "extract and store relevant content from sequential information chunks into an external memory system; organize stored content across core, episodic, and semantic memory components; correctly answer downstream questions using the full accumulated interaction history; invoke the right memory-operation tool (from a multi-tool memory architecture) at the right time",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Each new information chunk must be integrated into the existing core/episodic/semantic memory structure, so downstream QA accuracy depends on cumulative, correctly structured memory built from all previously processed chunks.",
  "n_goals": null,
  "tracking_demand": "The agent must track which information has been extracted and stored, how it is structured across core/episodic/semantic memory, and how it should be updated as new chunks arrive, since reward is downstream QA accuracy over the full interaction history.",
  "scoring": "other:downstream-QA-accuracy-over-full-history. The reward/evaluation signal derives from downstream question-answering accuracy over the full interaction history rather than per-chunk partial credit, though this implicitly requires correct memory construction across many intermediate chunks.",
  "horizon_value": "trained up to 30k tokens; generalizes to sequences exceeding 400k tokens (>13x training length)",
  "horizon_unit": "other:tokens-of-accumulated-interaction-history",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the full multi-chunk interaction/context processed before downstream QA)",
  "horizon_stated": "yes",
  "headline_result": "Mem-alpha achieves significant improvements over existing memory-augmented agent baselines and, despite training on sequences up to 30k tokens, generalizes to sequences exceeding 400k tokens (over 13x the training length); no human/expert baseline given.",
  "availability": null,
  "goal_span": "During training, agents process sequential information chunks, learn to extract and store relevant content, then update the memory system. The reward signal derives from downstream question-answering accuracy over the full interaction history, directly optimizing for memory construction.",
  "horizon_span": "Despite being trained exclusively on instances with a maximum length of 30k tokens, our agents exhibit remarkable generalization to sequences exceeding 400k tokens, over 13x the training length, highlighting the robustness of Mem-alpha.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281682069",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280136659",
  "title": "Evaluating Memory in LLM Agents via Incremental Multi-Turn Interactions",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 213,
  "publication_date": "2025-07-07",
  "months_since_pub": 14,
  "citations_per_month": 15.21,
  "artifact_name": "MemoryAgentBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "accurately retrieve previously seen information from accumulated context; adapt to and learn from new information at test time; understand and reason over long-range accumulated context; selectively forget information that is no longer relevant or valid",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Each new turn incrementally adds information that must be integrated into existing memory, so later queries testing accurate retrieval, test-time learning, or long-range understanding depend on correctly having accumulated (and, for selective forgetting, correctly discarded) information from all earlier turns.",
  "n_goals": "four core memory competencies (accurate retrieval, test-time learning, long-range understanding, selective forgetting)",
  "tracking_demand": "The memory agent must incrementally accumulate, update, and retrieve information across multi-turn interactions, and correctly discard information rendered obsolete while retaining what remains useful.",
  "scoring": "other:per-competency-accuracy. The benchmark reports separate scores per each of the four core competencies -- a per-goal/subgoal-type breakdown -- rather than a single undifferentiated final-episode score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Current memory agents (from simple context/RAG systems to advanced external-memory and tool-using agents) fall short of mastering all four competencies simultaneously; no specific numeric top-line figure or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "we identify four core competencies essential for memory agents: accurate retrieval, test-time learning, long-range understanding, and selective forgetting... Our benchmark transforms existing long-context datasets and incorporates newly constructed datasets into a multi-turn format, effectively simulating the incremental information processing characteristic of memory agents.",
  "horizon_span": "Our benchmark transforms existing long-context datasets and incorporates newly constructed datasets into a multi-turn format, effectively simulating the incremental information processing characteristic of memory agents.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280136659",
  "provenance": "asta-find,parametric,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282210434",
  "title": "MemoryBench: A Benchmark for Memory and Continual Learning in LLM Systems",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 49,
  "publication_date": "2025-10-20",
  "months_since_pub": 11,
  "citations_per_month": 4.45,
  "artifact_name": "MemoryBench",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "learn from accumulated user feedback received during service time; apply continually-updated knowledge correctly across multiple domains, languages, and task types; outperform static (non-continual-learning) baselines after repeated feedback exposure",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later performance depends on correctly incorporating feedback accumulated from earlier interactions within the same simulated user-feedback stream, so failing to integrate earlier feedback degrades later-task performance.",
  "n_goals": null,
  "tracking_demand": "The system must track and integrate a stream of accumulated user feedback over service time, across multiple domains, languages, and task types, updating its internal state/parameters rather than treating each query independently.",
  "scoring": "other:continual-learning-effectiveness-and-efficiency-vs-baselines",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "State-of-the-art baselines are reported as 'far from satisfying' in both effectiveness and efficiency; no specific top score or human-expert comparison is given in the abstract.",
  "availability": "https://memorybench.thuir.cn (website); https://github.com/THUIR/MemoryBench (code); https://huggingface.co/datasets/THUIR/MemoryBench and MemoryBench-Full (data)",
  "goal_span": "we propose a user feedback simulation framework and a comprehensive benchmark covering multiple domains, languages, and types of tasks to evaluate the continual learning abilities of LLMsys. Experiments show that the effectiveness and efficiency of state-of-the-art baselines are far from satisfying",
  "horizon_span": "testing their abilities to learn from accumulated user feedback in service time",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282210434",
  "provenance": "forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "277955197",
  "title": "Know Me, Respond to Me: Benchmarking LLMs for Dynamic User Profiling and Personalized Responses at Scale",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 158,
  "publication_date": "2025-04-19",
  "months_since_pub": 17,
  "citations_per_month": 9.29,
  "artifact_name": "PERSONAMEM",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "internalize a user's inherent traits and preferences from interaction history; track how the user's profile/preferences evolve over time across sessions; generate a personalized response consistent with the current (most up-to-date) state of the user's profile in a new scenario",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Because user traits/preferences evolve across up to 60 sessions, correctly answering a later in-situ query depends on having tracked and updated the profile from all relevant earlier sessions, so using a stale (outdated) profile state produces a misaligned response.",
  "n_goals": "over 180 simulated user-LLM interaction histories, each with up to 60 sessions, across 15 real-world personalization tasks",
  "tracking_demand": "The chatbot must internalize inherent user traits, track how the user's profile evolves session by session, and select the response consistent with the user's current (not stale) profile state when answering an in-situ, first-person query.",
  "scoring": "other:overall-response-selection-accuracy",
  "horizon_value": "over 180 simulated interaction histories, each containing up to 60 sessions of multi-turn conversations",
  "horizon_unit": "sessions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/history (a single simulated user-LLM interaction history, up to 60 sessions long)",
  "horizon_stated": "yes",
  "headline_result": "Frontier models such as GPT-4.1, o4-mini, GPT-4.5, o1, and Gemini-2.0 achieve only around 50% overall accuracy at selecting the response matching the user's current profile state; no human/expert baseline is given.",
  "availability": "github.com/bowen-upenn/PersonaMem",
  "goal_span": "PERSONAMEM features curated user profiles with over 180 simulated user-LLM interaction histories, each containing up to 60 sessions of multi-turn conversations across 15 real-world tasks that require personalization... frontier models such as GPT-4.1, o4-mini, GPT-4.5, o1, or Gemini-2.0 achieving only around 50% overall accuracy",
  "horizon_span": "PERSONAMEM features curated user profiles with over 180 simulated user-LLM interaction histories, each containing up to 60 sessions of multi-turn conversations across 15 real-world tasks that require personalization.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277955197",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "sessions",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282272250",
  "title": "Beyond Reactivity: Measuring Proactive Problem Solving in LLM Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": true,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 8,
  "publication_date": "2025-10-22",
  "months_since_pub": 11,
  "citations_per_month": 0.73,
  "artifact_name": "PROBE",
  "artifact_kind": "benchmark",
  "domain": "personal-assistant-memory",
  "goal_types": "search for unspecified, unprompted issues across the user's available context/data; identify the specific bottleneck underlying an ambiguous problem; execute an appropriate resolution action autonomously",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "sequential-chain",
  "interdependence": "The three capabilities form a strict pipeline: correct execution depends on having correctly identified the bottleneck, which in turn depends on having found the relevant unspecified issue via search, so an error early in the chain propagates forward through the task.",
  "n_goals": null,
  "tracking_demand": "The agent must track candidate evidence surfaced during open-ended search, determine which issues are genuine unresolved bottlenecks, and carry that determination forward into selecting and executing a resolution, all without being told what to look for.",
  "scoring": "other:pipeline-based-evaluation. The abstract reports a single end-to-end performance figure (40% for the best models) plus per-model/per-failure-mode analysis, but does not explicitly describe a formal subgoal-checkpoint partial-credit scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best end-to-end performance of 40%, achieved by both GPT-5 and Claude Opus-4.1; no human baseline given in the abstract.",
  "availability": null,
  "goal_span": "PROBE decomposes proactivity as a pipeline of three core capabilities: (1) searching for unspecified issues, (2) identifying specific bottlenecks, and (3) executing appropriate resolutions.",
  "horizon_span": "current benchmarks are constrained to localized context, limiting their ability to test reasoning across sources and longer time horizons.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282272250",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "281675361",
  "title": "SimuHome: A Temporal- and Environment-Aware Benchmark for Smart Home LLM Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 7,
  "publication_date": "2025-09-29",
  "months_since_pub": 12,
  "citations_per_month": 0.58,
  "artifact_name": "SimuHome",
  "artifact_kind": "environment/simulator",
  "domain": "personal-assistant-memory",
  "goal_types": "answer state-inquiry questions about the current smart-home environment; infer implicit user intent behind an ambiguous request; execute explicit device-control commands via SimuHome APIs; schedule and coordinate multi-device workflows whose effects evolve environmental variables over time; recognize and appropriately reject infeasible requests",
  "goal_origin": "mixed:given-up-front-plus-emitted-by-environment",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Device operations continuously affect environmental variables (e.g. temperature, humidity), so scheduling one device's workflow changes the state that later commands/workflows must react to, creating temporal, state-based dependencies between device-control subgoals.",
  "n_goals": "600 episodes; 18 agents evaluated across 4 task categories (state inquiry, intent inference, device control, workflow scheduling)",
  "tracking_demand": "The agent must track how already-issued device commands change environmental variables over (accelerated) simulated time and use this state to judge whether new requests are feasible or already satisfied.",
  "scoring": "other:per-category-accuracy. The benchmark separately scores state inquiry, implicit-intent inference, explicit device control, and workflow scheduling, each with feasible and infeasible request variants, but the abstract does not describe a single subgoal-checkpoint credit scheme spanning categories.",
  "horizon_value": "600 episodes; simulator accelerates time so scheduled workflows can be evaluated immediately",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "across the whole benchmark (600 episodes total); each episode's real-world-equivalent duration is compressed via time acceleration",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "Our benchmark covers state inquiry, implicit user intent inference, explicit device control, and workflow scheduling, each with both feasible and infeasible requests... An evaluation of 18 agents reveals that workflow scheduling is the hardest category, with failures persisting across alternative agent frameworks and fine-tuning.",
  "horizon_span": "We introduce SimuHome, a high-fidelity smart home simulator and a benchmark of 600 episodes for LLM-based smart home agents... For workflow scheduling, the simulator accelerates time so that scheduled workflows can be evaluated immediately.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281675361",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280337739",
  "title": "UserBench: An Interactive Gym Environment for User-Centric Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "personal-assistant-memory",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "personal-assistant-memory",
  "citation_count": 62,
  "publication_date": "2025-07-29",
  "months_since_pub": 14,
  "citations_per_month": 4.43,
  "artifact_name": "UserBench",
  "artifact_kind": "environment/simulator",
  "domain": "personal-assistant-memory",
  "goal_types": "proactively clarify a simulated user's underspecified/vague initial goal; incrementally uncover and track multiple user preferences revealed gradually over a multi-turn interaction; make grounded decisions with tools that align with all of the user's (eventually revealed) intents",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Preferences are revealed incrementally rather than all at once, so full alignment requires the agent to proactively elicit each one across multiple turns; the abstract reports current agents uncover fewer than 30% of all user preferences through active interaction, showing preference discovery does not happen atomically.",
  "n_goals": null,
  "tracking_demand": "The agent must track which of the user's underspecified goals and incrementally revealed preferences it has already clarified, and continue to proactively elicit unclarified ones across the multi-turn interaction while using tools to act on what it has learned so far.",
  "scoring": "other:full-intent-alignment-rate-plus-preference-uncover-rate",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Models provide answers that fully align with all user intents only 20% of the time on average, and even the most advanced models uncover fewer than 30% of all user preferences through active interaction; no explicit human baseline is given (this compares leading open- and closed-source LLMs against each other).",
  "availability": null,
  "goal_span": "For instance, models provide answers that fully align with all user intents only 20% of the time on average, and even the most advanced models uncover fewer than 30% of all user preferences through active interaction.",
  "horizon_span": "UserBench features simulated users who start with underspecified goals and reveal preferences incrementally, requiring agents to proactively clarify intent and make grounded decisions with tools.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280337739",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291185159",
  "title": "ASI-Bench: At the Dawn of Artificial Superintelligence",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 2,
  "publication_date": "2026-08-18",
  "months_since_pub": 1,
  "citations_per_month": 2.0,
  "artifact_name": "ASI-Bench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "independently select an appropriate research method for a project-level research task; conduct the research (experimentation/analysis) using the selected method; produce verifiable results at the level of a full research project, with progressively less methodological guidance provided across three guidance tiers",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Each of the 60 project-level tasks requires method selection to precede research execution, which must precede producing verifiable results; withdrawing guidance at the method-selection stage compounds into larger failures at execution and results stages, hence the sharp score drop across guidance tiers.",
  "n_goals": "60 project-level research tasks across 11 scientific domains, each assessed at three progressively-less-guided tiers",
  "tracking_demand": "System must track which methodological guidance tier is currently active for a task, whether a method has been selected, and whether the resulting research output is verifiable against expert review, across an entire project-level research process.",
  "scoring": "milestone-rubric - all tasks undergo expert review, AI-assisted auditing, sandbox execution, and scorer validation, reported as a project-level average score that sharply declines (50.91 -> 29.10 -> 26.62) as methodological guidance is withdrawn; a graded, not simply binary, project-level scoring scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Across 18 state-of-the-art agent-model configurations, the average score drops from 50.91 with full methodological guidance to 29.10 with only the method specified and 26.62 when agents must determine the method themselves; no human/expert numeric baseline is given beyond the expert-built task difficulty itself.",
  "availability": "https://asibench.apexin.ai/submit",
  "goal_span": "ASI-Bench contains 60 project-level research tasks across 11 scientific domains and progressively reduces methodological guidance to test whether AI can independently select methods, conduct research, and produce verifiable results.",
  "horizon_span": "Built by over 40 experts with the cost of 31,000+ human hours, ASI-Bench contains 60 project-level research tasks across 11 scientific domains.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291185159",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284860495",
  "title": "AstroReason-Bench: Evaluating Unified Agentic Planning across Heterogeneous Space Planning Problems",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-01-16",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "AstroReason-Bench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "schedule ground-station communication passes; schedule agile Earth-observation tasks; satisfy heterogeneous mission objectives under strict physical constraints within a Space Planning Problem instance",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Ground-station communication windows and Earth-observation opportunities compete for shared satellite/scheduling resources under strict physical constraints, so committing to one objective can preclude others.",
  "n_goals": null,
  "tracking_demand": "Agent must track scheduling state across multiple heterogeneous objectives (communication windows, observation opportunities) and physical/orbital constraints simultaneously.",
  "scoring": "other:comparison-against-specialized-solvers",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "AstroReason-Bench integrates multiple scheduling regimes, including ground station communication and agile Earth observation, and provides a unified agent-oriented interaction protocol.",
  "horizon_span": "a family of high-stakes problems with heterogeneous objectives, strict physical constraints, and long-horizon decision-making",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284860495",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291367768",
  "title": "BixBench3: Benchmarking AI agents on research-study-scale computational biology tasks",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 0,
  "publication_date": "2026-08-26",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "BixBench3",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "execute a sequence of computational-biology analyses from raw data through to a research objective; produce each of multiple data artifacts (e.g., peak call matrices, differential expression tables) matching the original published study; manage large raw datasets (some over 100GB) within time/cost constraints; maintain coherence across multiple sequential analysis steps",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later analyses depend on artifacts produced by earlier steps in the same sequence (e.g., differential-expression analysis depends on earlier peak calling/normalization), and agents perform worse on analyses requiring more sequential steps (0.36 average score at 1-2 steps vs. 0.24 at 3+ steps).",
  "n_goals": "20 tasks encompassing generation of 138 unique artifacts",
  "tracking_demand": "The agent must track intermediate data artifacts produced at each analysis step, manage large raw datasets, and maintain coherence across multiple sequential analyses before producing final gradable artifacts.",
  "scoring": "subgoal-checkpoint-partial-credit. Each resulting data artifact (e.g., peak call matrix, differential expression table) is programmatically graded against the corresponding artifact from the original published study, giving explicit per-artifact partial credit across the 138 unique artifacts rather than one final binary score.",
  "horizon_value": "average 6.8 hours per task (longest attempts up to 24 hours)",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "13 frontier models score from 0.00 (Gemini 3.1 Flash Lite) to 0.48 (GPT 5.6 Sol); no human/expert baseline given, though scores clearly fall short of the original published studies' own results.",
  "availability": null,
  "goal_span": "The data artifacts resulting from these analyses - such as peak call matrices or differential expression tables - are programmatically graded against the corresponding artifacts generated and reported in the original study. Across 20 BixBench3 tasks encompassing the generation of 138 unique artifacts, we find that 13 frontier models achieve scores ranging from 0.00 for Gemini 3.1 Flash Lite to 0.48 for GPT 5.6 Sol.",
  "horizon_span": "On average, agents use 6.8 hours, 102 million tokens, and $43 to complete each task, with the longest attempts consuming 24 hours, 1.07 billion tokens, and $525.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291367768",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288110638",
  "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-05-08",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "ChemCost",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "ground chemical identities from a reaction description; retrieve supplier quotes for the grounded chemicals; select valid purchasable packs matching required quantities; normalize quantities across packs; compute the total procurement cost from the reaction description",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each stage depends on the correctness of the previous one -- an incorrect chemical grounding invalidates supplier-quote retrieval, which invalidates pack selection, quantity normalization, and the final cost computation -- and the paper explicitly supports stage-level diagnosis of exactly this cascade (grounding, retrieval, procurement, arithmetic failures).",
  "n_goals": "5 chained sub-tasks (ground, retrieve, select, normalize, compute) per reaction; 1,427 evaluable reactions",
  "tracking_demand": "The agent must track which chemicals have been correctly grounded, which supplier quotes and packs have been retrieved/selected, and carry normalized quantities forward correctly into the final arithmetic cost computation, especially under noise-injected perturbations (aliases, quantity expressions, missing fields, formatting).",
  "scoring": "other:scalar-accuracy-within-relative-error-threshold-plus-stage-level-diagnosis",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Strongest agents reach only 50.6% accuracy within 25% relative error on clean inputs and degrade substantially with realistic noise; no human baseline given.",
  "availability": null,
  "goal_span": "we address this gap with chemical procurement cost estimation, a practical task in which an agent must ground chemical identities, retrieve supplier quotes, select valid purchasable packs, normalize quantities, and compute cost from a reaction description. We introduce ChemCost, a benchmark of 1,427 evaluable reactions grounded to a frozen pricing snapshot ... supporting scalar scoring and stage-level diagnosis of grounding, retrieval, procurement, and arithmetic failures.",
  "horizon_span": "We introduce ChemCost, a benchmark of 1,427 evaluable reactions grounded to a frozen pricing snapshot covering 2,261 chemicals and 230,775 supplier quotes",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288110638",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288669908",
  "title": "DiscoverPhysics: Benchmarking LLMs for Out-of-the-Box Scientific Thinking",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 4,
  "publication_date": "2026-05-25",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "DiscoverPhysics",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "design a sequence of informative experiments to probe an unknown simulated world's physics; revise hypotheses about the governing physical law across multiple rounds based on observed trajectory data; submit both a natural-language explanation and a Python implementation of the inferred law for each of 22 worlds",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "Each round of experiment design and hypothesis revision builds on evidence from prior rounds, so later experiments and hypotheses are conditioned on what earlier rounds revealed about the world's hidden physics.",
  "n_goals": "22 worlds, each requiring several rounds of experiments",
  "tracking_demand": "Agent must track its accumulated experimental observations (trajectory data) and current hypothesis about the world's physics across several rounds before submitting a final explanation and implementation.",
  "scoring": "other:trajectory-mse-plus-llm-judged-explanation - scored along two axes (trajectory MSE on held-out particles, and an LLM-judged explanation score against an expert rubric); abstract does not describe intermediate per-round subgoal checkpoint credit, only final-explanation/implementation quality.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Across eleven frontier models, the strongest agents pass only about half of the 22 worlds; open-source models lag substantially behind commercial ones; no explicit human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Each world is generated on demand by an N-body simulator, for which the agent proposes several rounds of experiments, observes raw trajectory data, and ultimately submits both a natural-language explanation of the world's physics and a Python implementation of the inferred law.",
  "horizon_span": "the agent proposes several rounds of experiments, observes raw trajectory data, and ultimately submits both a natural-language explanation of the world's physics and a Python implementation of the inferred law.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288669908",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289097517",
  "title": "InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-06-08",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "InquiTree (IT-18 subset)",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "formulate a hypothesis consistent with a paper-derived research tree's logical dependencies; design a study to test the hypothesis; interpret the resulting study outcome; update beliefs/conclusions based on interpreted results, propagating correctly through the DAG",
  "goal_origin": "mixed:tree-structure-given-agent-generates-navigation-decisions",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Research-tree nodes encode logical dependencies among hypothesis formulation, study design, interpretation, and belief updating, so an error or omission at an earlier node propagates and constrains which downstream inquiry paths remain valid.",
  "n_goals": "30-paper test pool; IT-18 subset with 120 listed subtopics (H=4 max hints), yielding a theoretical range of 360-1320 reasoning-action turns for full traversal",
  "tracking_demand": "Agents must track their evolving beliefs/conclusions across nested subtopic branches, remain consistent with prior hypothesis/study/interpretation nodes, and avoid 'Erosion of Marginal Capabilities' (degrading critical judgment) over long-horizon interactions.",
  "scoring": "milestone-rubric -- performance is diagnosed through the interactive Research Tree's node-level dependencies (hypothesis, study design, interpretation, belief updating) rather than a single end-of-episode pass/fail, though the abstract does not name an explicit points-based rubric.",
  "horizon_value": "theoretical interaction range of roughly 360 (3x120) to 1320 (11x120) reasoning-action turns for full traversal of the IT-18 subset (120 subtopics, H=4); per-task bounds of 21-77 steps for a typical task (n=7, H=4)",
  "horizon_unit": "turns",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "both -- per-task bounds (21-77 steps for a typical n=7 task) and a full-benchmark-traversal range (360-1320 turns across 120 subtopics)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce InquiTree, a diagnostic environment that formalizes scientific inquiry as interactive Research Trees: directed acyclic graphs capturing the logical dependencies among hypothesis formulation, study design, result interpretation, and belief updating.",
  "horizon_span": "With H=4 and 120 listed subtopics in the released IT-18 subset, this yields a theoretical interaction range of roughly 3\u00d7120=360 to 11\u00d7120=1320 reasoning-action turns for full traversal, placing the benchmark firmly in the long-horizon regime.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289097517",
  "provenance": "asta-find",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289299785",
  "title": "LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-06-15",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "LabOSBench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "complete each stage of a scientific-instrument operation workflow: sample loading, alignment, parameter tuning, data acquisition, and result inspection; perform feedback-driven parameter adjustment based on live instrument readouts; correctly operate one of 8 distinct instrument simulators across 96 subtasks",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Instrument workflows are staged (sample loading -> alignment -> parameter tuning -> data acquisition -> result inspection), so later stages require correct completion and calibration from earlier stages; feedback-driven parameter tuning also requires the agent to respond to instrument readouts generated by its own prior actions.",
  "n_goals": "96 subtasks across 8 instrument simulators",
  "tracking_demand": "Agent must track which workflow stage it is in (loading, alignment, tuning, acquisition, inspection), instrument readouts/feedback from its own prior actions, and calibration state carried from earlier stages.",
  "scoring": "other:subtask-plus-end-to-end-evaluation - evaluates general-purpose VLMs, specialized GUI agent models, and agentic frameworks at both subtask and end-to-end levels; an explicit subtask-level (subgoal) evaluation in addition to end-to-end scoring.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Existing agents can complete many structured GUI subtasks but still struggle with feedback-driven operations and long-horizon workflow execution; no specific numeric top score or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "LabOSBench constructs 96 subtasks across eight instrument simulators, covering workflows from sample loading, alignment, parameter tuning, and data acquisition to result inspection.",
  "horizon_span": "Our experiments reveal that while existing agents can complete many structured GUI subtasks, they still struggle with feedback-driven operations and long-horizon workflow execution.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289299785",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291266567",
  "title": "LifeSciBench: Evaluating Language Models on Realistic, Expert-Level Tasks in the Life Sciences",
  "year": 2026,
  "venue": "bioRxiv",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 4,
  "publication_date": "2026-08-18",
  "months_since_pub": 1,
  "citations_per_month": 4.0,
  "artifact_name": "LifeSciBench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "execute a chain of multiple dependent judgment calls within one realistic life-science research task; satisfy a human-expert-written rubric spanning one of seven representative scientific workflows; operate correctly across each of seven life-science domains",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks require the accurate execution of multiple dependent judgment calls, meaning an error in an earlier judgment call propagates into rubric failure on later, dependent parts of the same task.",
  "n_goals": "750 expert-authored tasks across seven workflows and seven domains, each with its own expert rubric",
  "tracking_demand": "The agent must track which dependent judgment calls it has made so far within a task and ensure later calls remain consistent with earlier ones, as graded against a human expert-written rubric.",
  "scoring": "milestone-rubric",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "GPT-Rosalind performs best with a task-weighted mean normalized rubric score of 0.576 and pass rate of 36.1%; 171 tasks (22.8%) have no passing response from any model; no human baseline given.",
  "availability": null,
  "goal_span": "LifeSciBench addresses this gap by spanning seven representative scientific workflows and seven life science domains, with each constituent task paired with a human expert-written rubric. ... which often involves ambiguities and requires the accurate execution of multiple dependent judgment calls.",
  "horizon_span": "We introduce LifeSciBench, a benchmark of 750 expert-authored tasks designed to evaluate whether language models can handle realistic life science research work.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291266567",
  "provenance": "web-registry",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286774610",
  "title": "Can LLM Agents Generate Real-World Evidence? Evaluating Observational Studies in Medical Databases",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": "healthcare-clinical",
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 2,
  "publication_date": "2026-03-24",
  "months_since_pub": 6,
  "citations_per_month": 0.33,
  "artifact_name": "RWE-bench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "construct a patient cohort from a real database (MIMIC-IV) per a study protocol; perform an analysis matching a peer-reviewed observational study's methodology; produce a coherent, tree-structured evidence bundle for reporting; iteratively execute and refine experiments against the reference protocol",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Cohort construction constrains what data is available for analysis, and analysis results constrain what can be validly reported, so errors in early cohort-construction steps propagate through the tree-structured evidence bundle; an automated cohort-evaluation method is used to localize such errors.",
  "n_goals": "162 tasks",
  "tracking_demand": "The agent must track its cohort definition, intermediate analysis results, and their organization into a tree-structured evidence bundle, checking internal coherence against the reference study protocol across the whole task.",
  "scoring": "other:question-level-plus-end-to-end-task-metrics. The paper evaluates using both question-level correctness and end-to-end task metrics, a form of subgoal/step-level partial credit in addition to overall task success (best agent 39.9%, best open-source 30.4%).",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Best agent reaches 39.9% task success and the best open-source model reaches 30.4% across 162 tasks; agent scaffold choice causes over 30% variation in performance; no human/expert baseline explicitly given.",
  "availability": "https://github.com/somewordstoolate/RWE-bench",
  "goal_span": "Each task provides the corresponding study protocol as the reference standard, requiring agents to execute experiments in a real database and iteratively generate tree-structured evidence bundles.",
  "horizon_span": "requiring agents to execute experiments in a real database and iteratively generate tree-structured evidence bundles",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286774610",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285607246",
  "title": "SciAgentGym: Benchmarking Multi-Step Scientific Tool-use in LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 9,
  "publication_date": "2026-02-13",
  "months_since_pub": 7,
  "citations_per_month": 1.29,
  "artifact_name": "SciAgentGym / SciAgentBench",
  "artifact_kind": "environment/simulator",
  "domain": "scientific-discovery",
  "goal_types": "orchestrate domain-specific scientific tools correctly across four natural science disciplines; complete tasks across a tiered difficulty spectrum from elementary actions to long-horizon workflows; sustain performance as interaction horizons extend rather than degrading",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "SciForge synthesizes trajectories by modeling the tool action space as a dependency graph, implying that later scientific tool calls in a workflow depend on outputs/preconditions established by earlier tool calls within the same discipline's workflow.",
  "n_goals": "1,780 domain-specific tools across four natural science disciplines; a tiered evaluation suite spanning elementary actions to long-horizon workflows",
  "tracking_demand": "The agent must track which of 1,780 domain-specific tools it has invoked and their outputs across a tiered workflow, since performance is shown to degrade substantially as the interaction horizon extends, requiring sustained state-tracking rather than one-off calls.",
  "scoring": "other:tiered-stress-test-from-elementary-to-long-horizon-with-degradation-analysis",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The authors' fine-tuned SciAgent-8B outperforms the much larger Qwen3-VL-235B-Instruct and shows positive cross-domain transfer; state-of-the-art models generally show a critical bottleneck with substantial performance degradation as interaction horizons extend. No explicit human/expert baseline is given.",
  "availability": null,
  "goal_span": "we present SciAgentBench, a tiered evaluation suite designed to stress-test agentic capabilities from elementary actions to long-horizon workflows. Our evaluation identifies a critical bottleneck: state-of-the-art models still struggle with complex scientific tool-use, and their performance degrades substantially as interaction horizons extend.",
  "horizon_span": "their performance degrades substantially as interaction horizons extend",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285607246",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289706430",
  "title": "Autonomous Scientific Discovery via Iterative Meta-Reflection",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 1,
  "publication_date": "2026-07-01",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "iNatDisco",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "generate scientific hypotheses about ecological patterns from a dataset without a pre-specified research question; validate each proposed hypothesis via statistical testing before accepting it; periodically synthesize accumulated prior discoveries to redirect exploration toward unexplored regions of the hypothesis space; incorporate multimodal tool use (e.g., image processing) to extract further supporting evidence",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "open-ended",
  "interdependence": "Each new round of hypothesis generation is conditioned on, and must remain statistically consistent with, the accumulated set of prior validated discoveries, since the meta-reflection step treats earlier findings as data constraining where to search next.",
  "n_goals": "9 known ground-truth patterns (8 recovered)",
  "tracking_demand": "The agent must maintain a growing record of prior discoveries and their statistical validation status, and periodically re-analyze that record to identify structural patterns, confounds, and epistemic gaps.",
  "scoring": "other:pattern-recovery-count-plus-hypothesis-support-rate. The abstract reports pattern-recovery count (8/9) and hypothesis support rate (72.7%) as its core metrics; no separate subgoal-level partial-credit scheme is described.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "DiscoPER recovers 8 of 9 known patterns with a 72.7% hypothesis support rate on iNatDisco, outperforming classical causal discovery and LLM-guided baselines; no human/expert baseline number is given in the abstract.",
  "availability": null,
  "goal_span": "Evaluated on iNatDisco, a new multimodal ecological knowledge benchmark with pattern-level ground truth obtained from peer-reviewed literature, DiscoPER recovers 8 of 9 known patterns with a 72.7% hypothesis support rate, outperforming both classical causal discovery and LLM-guided baselines.",
  "horizon_span": "DiscoPER recovers 8 of 9 known patterns with a 72.7% hypothesis support rate",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289706430",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282100257",
  "title": "Evaluating large language model agents for automation of atomic force microscopy",
  "year": 2025,
  "venue": "Nature Communications",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 61,
  "publication_date": "2025-10-14",
  "months_since_pub": 11,
  "citations_per_month": 5.55,
  "artifact_name": "AFMBench",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "complete the full scientific workflow from experimental design to results analysis for atomic force microscopy (AFM) automation; successfully perform each of several increasingly advanced experiments: AFM calibration, feature detection, mechanical property measurement, graphene layer counting, and indenter detection; coordinate across multi-agent roles in laboratory settings without deviating from instructions ('sleepwalking')",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Later, more advanced experiments (e.g., mechanical property measurement, indenter detection) build on successfully completing earlier, more basic capabilities (e.g., calibration, feature detection), and coordination failures or instruction deviations at one stage propagate into downstream experimental steps.",
  "n_goals": "5 named experiment types of increasing difficulty (calibration, feature detection, mechanical property measurement, graphene layer counting, indenter detection)",
  "tracking_demand": "Agent(s) must track experimental state and configuration across the full workflow (design, calibration, measurement, analysis), coordinate with other agents in multi-agent setups, and avoid deviating from given instructions across a sequence of physical lab actions.",
  "scoring": "other:per-experiment-capability-evaluation - a comprehensive evaluation suite challenging agents 'across the complete scientific workflow'; the five named experiments of increasing difficulty imply per-experiment (subgoal-level) pass/fail rather than a single end-to-end binary score, though the abstract does not spell out a formal partial-credit rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "State-of-the-art LLMs struggle with basic tasks and coordination scenarios; models excelling at materials-science QA perform poorly in laboratory settings; multi-agent frameworks significantly outperform single-agent approaches; no explicit human/expert baseline is reported.",
  "availability": null,
  "goal_span": "we develop AFMBench\u2014a comprehensive evaluation suite challenging LLM agents across the complete scientific workflow from experimental design to results analysis... Finally, we evaluate AILA's effectiveness in increasingly advanced experiments\u2014AFM calibration, feature detection, mechanical property measurement, graphene layer counting, and indenter detection.",
  "horizon_span": "we develop AFMBench\u2014a comprehensive evaluation suite challenging LLM agents across the complete scientific workflow from experimental design to results analysis.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282100257",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "275324069",
  "title": "BoxingGym: Benchmarking Progress in Automated Experimental Design and Model Discovery",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 11,
  "publication_date": "2025-01-02",
  "months_since_pub": 20,
  "citations_per_month": 0.55,
  "artifact_name": "BoxingGym",
  "artifact_kind": "benchmark",
  "domain": "scientific-discovery",
  "goal_types": "design and run informative experiments that reduce uncertainty about a generative model's parameters; propose a scientific model/theory of the given environment; revise the proposed theory in light of newly collected experimental data; produce an explanation of the model that lets another agent make reliable predictions",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Revising the scientific theory depends on data collected from previously run experiments, and the informativeness of each new experiment (measured via expected information gain) depends on what has already been learned about the model's parameters.",
  "n_goals": "10 environments (drawn from real-world scientific domains from psychology to ecology)",
  "tracking_demand": "The agent must track what has been learned from experiments already run (to target expected information gain for the next experiment) and maintain an evolving explanation of its current best scientific model.",
  "scoring": "other:EIG-plus-explanation-based-prediction-accuracy. The abstract describes two evaluation channels -- expected information gain for experimental design, and explanation-enabled prediction accuracy for model discovery -- rather than a single milestone/subgoal-checkpoint scheme.",
  "horizon_value": "evaluated after 0, 1, 3, 5, 7, and 10 experiment-design steps per trial, with 5 independent trials per environment",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per trial (per environment); results averaged across 5 independent trials",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we implement each environment as a generative probabilistic model with which a scientific agent can run interactive experiments... we compute the expected information gain (EIG)... to quantitatively evaluate model discovery, we ask a scientific agent to explain their model and then assess whether this explanation enables another scientific agent to make reliable predictions.",
  "horizon_span": "At each step, the agent chooses to perform an experiment, by specifying a design, and observes the outcome. After a fixed number of steps (0, 1, 3, 5, 7, 10), we evaluate the agent's performance ... For each environment, we run the agents for 5 independent trials.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:275324069",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278912205",
  "title": "ScienceBoard: Evaluating Multimodal Autonomous Agents in Realistic Scientific Workflows",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "scientific-discovery",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "scientific-discovery",
  "citation_count": 43,
  "publication_date": "2025-05-26",
  "months_since_pub": 16,
  "citations_per_month": 2.69,
  "artifact_name": "ScienceBoard",
  "artifact_kind": "environment/simulator",
  "domain": "scientific-discovery",
  "goal_types": "autonomously interact with professional scientific software (biochemistry, astronomy, geoinformatics) to accomplish a research task; complete each of 169 rigorously validated real-world scientific-discovery workflow tasks",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Complex research tasks and experiments within the multi-domain environment require agents to interact via different interfaces across a workflow, implying later steps in a scientific workflow build on outputs (e.g. data/results) produced by earlier interface interactions, though the abstract does not spell out an explicit dependency graph.",
  "n_goals": "169 tasks across biochemistry, astronomy, and geoinformatics",
  "tracking_demand": "Agent must track intermediate results and state produced while autonomously interacting with dynamic, visually rich professional software across a multi-step scientific workflow.",
  "scoring": "binary-final-success \u2014 the abstract reports 'only a 15% overall success rate', with no mention of subgoal-level or partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Agents with state-of-the-art backbones (GPT-4o, Claude 3.7, UI-TARS) achieve only a 15% overall success rate (no human baseline given).",
  "availability": "https://qiushisun.github.io/ScienceBoard-Home/",
  "goal_span": "a challenging benchmark of 169 high-quality, rigorously validated real-world tasks curated by humans, spanning scientific-discovery workflows in domains such as biochemistry, astronomy, and geoinformatics",
  "horizon_span": "a challenging benchmark of 169 high-quality, rigorously validated real-world tasks curated by humans",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278912205",
  "provenance": "forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291531459",
  "title": "APIFlow-Bench: Measuring Whether Agents Survive Long, Dependent API Workflows",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-08-29",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "APIFlow-Bench",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "complete each subtask in a long, dependent chain of REST-API calls correctly (state, not just completion, must be right); produce a final answer whose delivery is traceable to the actual call path (provenance-sensitive correctness); avoid compounding failures across increasingly long dependency chains (up to 20 subtasks)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Subtasks are generated 'forward, subtask by subtask', each admitted only after its grader is verified solvable, so later subtasks in a chain causally depend on the state produced by earlier ones; a canary traced through the API data flow ties final delivery to the whole prior call path.",
  "n_goals": "up to 20 subtasks per chain (across 19 evaluated models, one neutral scaffold)",
  "tracking_demand": "Agent must track state correctness at each subtask in a dependency chain (not just whether the chain 'completed'), including whether a mock-minted canary correctly propagates through the API data flow to the final delivered answer.",
  "scoring": "other:mixed \u2014 decomposes performance into 'seven engineering capabilities' with deterministic, provenance-sensitive grading per subtask/chain (a state check plus a typed answer card verified field by field), which functions as subgoal-checkpoint-style partial credit distinct from a single completion bit.",
  "horizon_value": "20 (clean chains); up to 44,362 execution transcripts released",
  "horizon_unit": "other:subtasks-per-workflow-chain",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per workflow (episode = one dependency chain)",
  "horizon_stated": "yes",
  "headline_result": "Success degrades from 93% on individual subtasks to 74% on clean 20-subtask chains (61% including flagged trials); best-of-five reliability spans 44 points versus all-five-of-five (no human baseline given).",
  "availability": null,
  "goal_span": "We introduce APIFlow-Bench, a fully auditable benchmark for long-horizon, dependent REST-API workflows that decomposes performance into seven engineering capabilities and requires agents to produce answers supported by the actual call path.",
  "horizon_span": "longer dependency chains degrade success, from 93% on individual subtasks to 74% on clean 20-subtask chains and 61% when including the 8% of chain trials that a model-consensus screen flags as passed by no model",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291531459",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288147605",
  "title": "AgentEscapeBench: Evaluating Out-of-Domain Tool-Grounded Reasoning in LLM Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-05-08",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "AgentEscapeBench",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "invoke real external tool functions to satisfy a directed acyclic dependency graph over tools and items; track hidden state revealed incrementally through tool use; propagate intermediate results correctly across dependent tool calls to a deterministically verifiable final answer",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each task defines a DAG over tools and items, so a later tool call's validity/success depends on correctly having invoked and propagated results from its prerequisite tool calls earlier in the dependency graph.",
  "n_goals": "270 instances across five difficulty tiers (dependency-depth levels from 5 to 25)",
  "tracking_demand": "Agent must track hidden state revealed incrementally by tool calls, maintain clue adherence, and correctly propagate intermediate results through the dependency graph as depth increases.",
  "scoring": "other:success-rate-by-dependency-depth - reports success rate as a function of dependency-graph depth/difficulty tier (5 to 25), showing sharp performance drops as depth increases; a depth-stratified binary success measure rather than in-task partial credit.",
  "horizon_value": "difficulty tiers from 5 to 25 (dependency-graph depth levels)",
  "horizon_unit": "other:dependency-graph-depth-tiers",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task instance - each instance's difficulty tier fixes how deep its tool/item dependency chain runs",
  "horizon_stated": "yes",
  "headline_result": "Humans decline from 98.3% success at difficulty-5 to 80.0% at difficulty-25, while the best of sixteen evaluated models drops from 90.0% to 60.0% over the same range.",
  "availability": null,
  "goal_span": "Each task defines a directed acyclic dependency graph over tools and items, requiring agents to invoke real external functions, track hidden state revealed incrementally, propagate intermediate results, and submit a deterministically verifiable final answer.",
  "horizon_span": "Experiments with sixteen LLM agents and human participants show that performance drops sharply as dependency depth increases: humans decline from 98.3% success at difficulty-5 to 80.0% at difficulty-25, while the best model drops from 90.0% to 60.0%.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288147605",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287929799",
  "title": "AgentFloor: How Far Up the tool use Ladder Can Small Open-Weight Models Go?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-05-01",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "AgentFloor",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "follow instructions correctly (lowest tier); use tools correctly (mid-lower tier); coordinate multiple steps/tool calls together (mid-upper tier); sustain long-horizon planning under persistent constraints over many steps (top tier)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "The six-tier capability ladder is explicitly hierarchical (instruction following -> tool use -> multi-step coordination -> long-horizon planning), and the abstract notes the 'gap appears most clearly on long-horizon planning tasks that require sustained coordination and reliable constraint tracking over many steps', implying higher tiers require successfully sustaining what lower tiers test, extended over time.",
  "n_goals": "30 tasks organized into a six-tier capability ladder",
  "tracking_demand": "Agent must sustain constraint tracking and coordination reliably over many steps to succeed at the highest tiers, where 'neither side reaches strong reliability' even among frontier models.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports 16,542 scored runs across 30 tasks/six tiers but does not describe an explicit per-tier or per-step partial-credit formula beyond the tiered task structure itself.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The strongest open-weight model matches GPT-5 in aggregate on AgentFloor while being substantially cheaper/faster, but a gap remains on long-horizon planning tasks where 'neither side reaches strong reliability' (no human baseline given).",
  "availability": "benchmark, harness, sweep configurations, and full run corpus released (per abstract); no specific URL given.",
  "goal_span": "We introduce AgentFloor, a deterministic 30-task benchmark organized as a six-tier capability ladder, spanning instruction following, tool use, multi-step coordination, and long-horizon planning under persistent constraints.",
  "horizon_span": "We evaluate 16 open-weight models, from 0.27B to 32B parameters, alongside GPT-5 across 16,542 scored runs",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287929799",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "289823439",
  "title": "AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-07-06",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "AgentGym2",
  "artifact_kind": "environment/simulator",
  "domain": "tool-use-API",
  "goal_types": "execute end-to-end real-world procedures without relying on pre-packaged tool interfaces; discover available tools via active exploration of the environment; compose discovered tools to solve previously unseen tasks; remain robust to noisy and underspecified task information",
  "goal_origin": "implied-by-constraints",
  "decomposition": "open-ended",
  "interdependence": "Because tools must be discovered rather than given, later composition and task-solving goals depend on the success of earlier exploration steps that reveal what tools/interfaces exist and how they behave under noise.",
  "n_goals": null,
  "tracking_demand": "Agent must track which tools/interfaces it has discovered so far, what it has verified about their (possibly noisy) behavior, and how these compose toward completing the current end-to-end task.",
  "scoring": "other:end-to-end-task-completion - abstract frames evaluation around whether agents can execute full end-to-end procedures; no explicit subgoal-checkpoint partial-credit scheme is described, though discovery/composition/robustness are separately called out as measured abilities.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Even SOTA systems like Gemini and GPT-5 struggle on AgentGym2, revealing a substantial gap between current agent capability and real-world demands; no specific numeric score is given in the abstract.",
  "availability": null,
  "goal_span": "Beyond reasoning and planning, it measures agents' ability to execute end-to-end procedures, discover tools via exploration, compose tools for unseen tasks, and remain robust to noisy and underspecified information.",
  "horizon_span": "To bridge this gap, we present AgentGym2, a new evaluation framework with task instances grounded in real-world end-to-end working demands.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289823439",
  "provenance": "forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290466796",
  "title": "AppWorld-UL: Benchmarking Diverse Agent-User Interactions for Tool-Use",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-07-10",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "AppWorld-UL",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "operate applications correctly to complete a digital task (e.g. ordering groceries) across 9 simulated apps; interact appropriately with the user: ask clarification questions, prompt for confirmation, or report infeasibility; succeed on compositional, multi-part sub-tasks that combine several of the above interaction types",
  "goal_origin": "mixed:given-up-front-task-with-user-clarification-injected-mid-episode",
  "decomposition": "hierarchical",
  "interdependence": "Compositional tasks require correctly resolved user interactions (clarification/confirmation/infeasibility) before or during app operation, so a missed or wrong clarification step degrades downstream task execution, which is why 'compositional' scores are much lower than the overall success rate.",
  "n_goals": "516 tasks, built on 9 simulated apps (e.g. Amazon, Spotify)",
  "tracking_demand": "Agent must track what has been asked/confirmed with the user so far, the user's carefully-bounded knowledge state (simulated by an LLM), and which parts of a compositional task remain to be completed.",
  "scoring": "other:mixed \u2014 reports an overall success rate, a harder compositional-subset success rate, and a stricter scenario-level metric, i.e. graded across multiple levels of strictness rather than a single binary outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Claude Opus 4.7 achieves only 48.6% success on AppWorld-UL, 35.7% on the harder compositional subset, and 21.3% on the stricter scenario-level metric (no human baseline given).",
  "availability": null,
  "goal_span": "we introduce AppWorld-UL, a ``user-in-the-loop'' benchmark of 516 challenging tasks requiring diverse agent-user interactions.",
  "horizon_span": "Building upon the AppWorld framework with 9 popular simulated apps like Amazon and Spotify",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290466796",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288701322",
  "title": "AsyncTool: Evaluating the Asynchronous Function Calling Capability under Multi-Task Scenarios",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 3,
  "publication_date": "2026-05-27",
  "months_since_pub": 4,
  "citations_per_month": 0.75,
  "artifact_name": "AsyncTool",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "concurrently manage multiple heterogeneous tasks presented simultaneously; make productive use of idle time while awaiting delayed tool-call responses (asynchronous tool calling); coordinate task switching, dependency tracking, and state maintenance across concurrently running tasks",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks run concurrently and share the agent's attention/idle time, so a decision to switch to a different task while waiting on one tool's delayed response affects when and how the original task's dependent next step can proceed, and interleaved task states must be tracked and kept consistent.",
  "n_goals": null,
  "tracking_demand": "Agent must track the state, dependencies, and pending tool responses of multiple simultaneously active tasks, and coordinate task-switching decisions during periods of delayed tool feedback.",
  "scoring": "other:step-subtask-task-level-efficiency-metrics - evaluates models at the step, sub-task, and task levels, and introduces efficiency-oriented metrics to measure task coordination and completion efficiency; explicit multi-level (step/sub-task/task) evaluation, i.e., subgoal-level credit exists.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Delayed tool feedback poses substantial challenges to current agents and leads to clear performance degradation; models that better coordinate task switching, dependency tracking, and state maintenance achieve stronger performance on AsyncTool; no specific numeric top score or human baseline is given.",
  "availability": null,
  "goal_span": "AsyncTool presents multiple heterogeneous tasks simultaneously and simulates realistic tool response latency during execution. Using a hybrid data evolution strategy, we construct a diverse asynchronous multitasking dataset that covers multiple scenarios and tool-use patterns. We evaluate models at the step, sub-task, and task levels, and introduce efficiency-oriented metrics to measure task coordination and completion efficiency.",
  "horizon_span": "AsyncTool presents multiple heterogeneous tasks simultaneously and simulates realistic tool response latency during execution.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288701322",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "291383111",
  "title": "Benchmarking AI Agents for Hardware Design Automation via MCP Tool Calling",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-08-25",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": null,
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "execute single-operation edits on hardware design components via specialised MCP tools; correctly sequence multi-step dependency chains (e.g. create component -> add port -> wire connection); handle invalid or misspelled requests without incorrect tool invocation; operate correctly across multi-server tool contexts",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Hardware design operations are dependency-ordered (a port must exist before it can be wired), so later tool calls in a chain are only valid once earlier dependency-establishing calls have succeeded; the benchmark explicitly separates single-operation edits from multi-step dependency chains by difficulty.",
  "n_goals": "1 expected call (easy) / 2 (medium) / 3-5 (hard) tool calls per task",
  "tracking_demand": "The agent must track which prior dependency-establishing tool calls (e.g. component creation, port creation) have already succeeded before issuing calls that depend on them, across single-agent and multi-agent tool-calling configurations.",
  "scoring": "other:expected-call-coverage-and-configuration-comparison",
  "horizon_value": "1 (Easy) / 2 (Medium) / 3-5 (Hard) expected tool calls per task",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task, by difficulty tier",
  "horizon_stated": "yes",
  "headline_result": "Strong models achieve near-complete expected-call coverage on benchmarked workflows, but reliability depends strongly on task structure and agent configuration; no single overall accuracy figure given in the abstract.",
  "availability": null,
  "goal_span": "engineers issue repetitive, dependency-ordered operations---such as creating components, adding ports, and wiring connections---through specialised tools. ... we build a Model Context Protocol (MCP) server ... and construct a benchmark covering single-operation edits, multi-step dependency chains, invalid requests, misspelled prompts, and multi-server tool contexts.",
  "horizon_span": "Easy, Medium, and Hard contain 40 self-contained tasks each ... Easy requires 1 expected call, Medium requires 2, and Hard requires 3-5 calls.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291383111",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284648439",
  "title": "C-World: A Computer Use Agent Environment Creator",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-01-09",
  "months_since_pub": 8,
  "citations_per_month": 0.12,
  "artifact_name": "C-World",
  "artifact_kind": "environment/simulator",
  "domain": "tool-use-API",
  "goal_types": "complete long-horizon workflows composed of many interacting constraints across up to 5,571 tools spanning 204 applications; correctly follow constraints despite injected realistic failures and perturbations during the task; satisfy a reward signal combining verifiable metrics with LLM-based judgment",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The Task Distribution engine synthesizes long-horizon workflows with 'wild constraints', and the Transition Function injects realistic failures/perturbations mid-task, so later steps must adapt to state changes and constraint violations introduced earlier in the same workflow.",
  "n_goals": null,
  "tracking_demand": "Agent must track constraint satisfaction across a long-horizon, multi-tool workflow while detecting and adapting to injected failures/perturbations introduced by the environment's transition function.",
  "scoring": "other:verifiable-metrics-plus-llm-judgment - Reward Signal combines verifiable metrics with LLM-based judgment; abstract identifies constraint following (not tool invocation) as the dominant failure mode, implying constraints are checked individually (subgoal-like) rather than only via one final aggregate score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Planning ability is uniformly strong across nine evaluated LLMs but execution remains the bottleneck; the World Engine achieves Spearman rho = 0.883 ranking correlation with real execution; fine-tuning on just 1,170 C-World trajectories outperforms baselines trained on 119k samples; no explicit human/expert baseline is given.",
  "availability": "https://ziqiao-git.github.io/C-World/",
  "goal_span": "We define a complete agent environment through four components: an Action Space of 5,571 format-unified tools across 204 common applications, a Task Distribution engine that synthesizes long-horizon workflows with wild constraints, a Transition Function implemented as a state controller that injects realistic failures and perturbations, and a Reward Signal combining verifiable metrics with LLM-based judgment.",
  "horizon_span": "a Task Distribution engine that synthesizes long-horizon workflows with wild constraints",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284648439",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286573213",
  "title": "CCTU: A Benchmark for Tool Use under Complex Constraints",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 7,
  "publication_date": "2026-03-16",
  "months_since_pub": 6,
  "citations_per_month": 1.17,
  "artifact_name": "CCTU",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "select and call the correct tool while satisfying every one of several simultaneous constraints (resource, behavior, toolset, response dimensions); maintain compliance with all constraints across a multi-turn interaction, not just the first tool call; self-refine after receiving feedback about a constraint violation",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Each test case involves an average of seven constraint types spanning resource, behavior, toolset, and response dimensions simultaneously, so satisfying one constraint can conflict with satisfying another, and 'an executable constraint validation module... performs step-level validation and enforces compliance during multi-turn interactions,' meaning compliance must hold at every step, not just the final one.",
  "n_goals": "200 test cases; 12 constraint categories across 4 dimensions (resource, behavior, toolset, response); average of 7 constraint types per test case",
  "tracking_demand": "The agent must track which of the ~7 simultaneous constraints (out of 12 categories across 4 dimensions) apply to the current tool-use scenario and re-check compliance with all of them at every step across the multi-turn interaction, especially after receiving feedback about a violation.",
  "scoring": "other:full-constraint-compliance-rate. 'No model achieves a task completion rate above 20%' when 'strict adherence to all constraints is required,' i.e. the paper measures full-compliance as an all-or-nothing per-task outcome, with separate analysis of violation rates by constraint dimension, but no formal per-constraint partial-credit rubric contributing to the headline score.",
  "horizon_value": "maximum 20 interaction rounds per test case",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per test case's multi-turn interaction)",
  "horizon_stated": "yes",
  "headline_result": "No model achieves a task completion rate above 20% under strict adherence to all constraints; models violate constraints in over 50% of cases overall, and self-refinement after feedback is limited. No human/expert baseline given.",
  "availability": null,
  "goal_span": "CCTU is grounded in a taxonomy of 12 constraint categories spanning four dimensions (i.e., resource, behavior, toolset, and response). The benchmark comprises 200 carefully curated and challenging test cases across diverse tool-use scenarios, each involving an average of seven constraint types and an average prompt length exceeding 4,700 tokens. To enable reliable evaluation, we develop an executable constraint validation module that performs step-level validation and enforces compliance during multi-turn interactions between models and their environments.",
  "horizon_span": "maximum 20 interaction rounds",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286573213",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287607601",
  "title": "GTA-2: Benchmarking General Tool Agents from Atomic Tool-Use to Open-Ended Workflows",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-04-17",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "GTA-2 (GTA-Atomic / GTA-Workflow)",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "execute short-horizon, closed-ended atomic tool calls correctly (GTA-Atomic); complete long-horizon, open-ended, real-world productivity workflows end-to-end (GTA-Workflow); satisfy verifiable sub-goals identified by a recursive checkpoint-based evaluation mechanism",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Open-ended GTA-Workflow objectives are recursively decomposed into verifiable sub-goals/checkpoints, so overall workflow completion depends on satisfying an ordered/nested set of checkpoints built from atomic tool-use capability.",
  "n_goals": null,
  "tracking_demand": "System must track which recursively-decomposed sub-goals/checkpoints of an open-ended workflow have been satisfied so far, across real deployed tools and multimodal contexts.",
  "scoring": "subgoal-checkpoint-partial-credit \u2014 explicitly, 'a recursive checkpoint-based evaluation mechanism that decomposes objectives into verifiable sub-goals', and the abstract notes 'checkpoint-guided feedback improves performance', confirming subgoal-level credit exists.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Frontier models already struggle on atomic tasks (below 50%) and largely fail on workflows, with top models achieving only 14.39% success on GTA-Workflow (no human baseline given).",
  "availability": "https://github.com/open-compass/GTA",
  "goal_span": "we propose a recursive checkpoint-based evaluation mechanism that decomposes objectives into verifiable sub-goals, enabling unified evaluation of both model capabilities and agent execution frameworks",
  "horizon_span": "(ii) GTA-Workflow introduces long-horizon, open-ended tasks for realistic end-to-end completion",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287607601",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287480765",
  "title": "GeoAgentBench: A Dynamic Execution Benchmark for Tool-Augmented Agents in Spatial Analysis",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-04-15",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "GeoAgentBench (GABench)",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "correctly configure parameters for each of 117 atomic GIS tools invoked; complete each of 53 typical spatial-analysis tasks across 6 core GIS domains; produce spatially/cartographically accurate outputs verified via a VLM-based check; decouple global workflow orchestration from step-wise reactive execution to recover from runtime anomalies",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Parameter choices at one tool-invocation step feed into subsequent tool calls in a multi-step GIS workflow, so a parameter misalignment or runtime anomaly at one step propagates into downstream execution failures unless recovered by the reactive layer.",
  "n_goals": "117 atomic GIS tools; 53 spatial-analysis tasks across 6 core GIS domains",
  "tracking_demand": "The agent must track its evolving execution plan, per-step parameter choices, and runtime feedback/anomalies across a multi-step GIS workflow to keep global orchestration consistent with step-wise reactive execution.",
  "scoring": "other:parameter-execution-accuracy-plus-VLM-verification. A dedicated Parameter Execution Accuracy (PEA) metric with a 'Last-Attempt Alignment' strategy plus VLM-based spatial/cartographic verification give explicit step/parameter-level partial credit alongside end-task correctness.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "GABench provides a realistic execution sandbox integrating 117 atomic GIS tools, encompassing 53 typical spatial analysis tasks across 6 core GIS domains.",
  "horizon_span": "GABench provides a realistic execution sandbox integrating 117 atomic GIS tools, encompassing 53 typical spatial analysis tasks across 6 core GIS domains.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287480765",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289668811",
  "title": "Ko-WideSearch: A Korean Breadth-Search Benchmark for Exhaustive Set Enumeration by Web Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-06-25",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "Ko-WideSearch",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "exhaustively enumerate the full membership of a named closed set (e.g. a TV season's cast, a dynasty's rulers); fill a per-item attribute table (multiple columns) for every enumerated member; decide when to stop searching within an open-ended web search space",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Item-, Column-, and Row-level correctness are scored jointly per table, so recovering the correct set of members (rows) without correctly filling every attribute column for each still yields incomplete credit; a fixed per-question iteration budget also creates a shared resource across all rows/columns still to be resolved.",
  "n_goals": "228 tables over 190 entities across sixteen categories",
  "tracking_demand": "The agent must track which set members it has already found (to avoid duplicates/omissions), which attribute columns remain unfilled per member, and its remaining search-iteration budget as difficulty knobs (table width, 2-D composite key) increase.",
  "scoring": "other:item-column-row-F1-per-table",
  "horizon_value": "fixed budget of 30 agent iterations per question (total tool calls run higher; one model logged up to 947 tool calls)",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/question",
  "horizon_stated": "yes",
  "headline_result": "Across twenty web agents, Item-F1 reaches 92.8 vs. Row-F1 only 53.7; accuracy falls as difficulty knobs harden and neither more search nor more spend closes the gap; no human baseline given.",
  "availability": null,
  "goal_span": "Each task names a set-parent entity ... and asks for its full membership plus a per-item attribute table, graded by Item-, Column-, and Row-F1. It spans 228 tables over 190 entities and sixteen categories across three difficulty tiers ... Across twenty web agents, the failure is consistent: agents recover the set but not the rows (e.g. Item-F1 92.8 against Row-F1 53.7)",
  "horizon_span": "each evaluated model receives ... a fixed per-question budget of thirty agent iterations, with the clarification that each iteration may batch several tool calls, so the total search-call count per task runs well above thirty. ... Qwen3.6-35B ... run 947 tool calls, the run maximum",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289668811",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288649984",
  "title": "TOBench: A Task-Oriented Omni-Modal Benchmark for Real-World Tool-Using Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-05-16",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "MM-ToolBench (TOBench)",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "execute tools appropriate to a Customer Service or Intelligent Creation task; inspect rendered or transformed intermediate artifacts produced by tool calls; self-correct when an inspected artifact fails task-specific requirements; satisfy each of the 20 subcategory slices' task-specific grounded evaluator checks",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The benchmark's central design is closed-loop multimodal verification: an agent must execute a tool, inspect the resulting artifact, and only then decide whether to self-correct, so each loop iteration's decision depends on the outcome of the previous tool execution and inspection.",
  "n_goals": "100 executable tasks across 20 subcategory slices, 27 MCP servers, 324 tools",
  "tracking_demand": "The agent must track the state of rendered/transformed artifacts across a closed verification loop, deciding when self-correction is required, using task-specific grounded evaluators across 27 MCP servers and 324 tools.",
  "scoring": "other:task-success-rate-vs-human-benchmark",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Claude Opus 4.6 achieves only 32.0% task success, far below the 94.0% human benchmark.",
  "availability": null,
  "goal_span": "The central design of MM-ToolBench is closed-loop multimodal verification: agents must execute tools, inspect rendered or transformed artifacts, and self-correct when outputs fail task-specific requirements. To make such evaluation scalable and verifiable, MM-ToolBench couples MCP-based execution with task-specific grounded evaluators",
  "horizon_span": "MM-ToolBench contains 100 executable tasks from two macro task families, Customer Service and Intelligent Creation, covering 20 subcategory slices and supported by 27 MCP servers with 324 tools.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288649984",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290143076",
  "title": "MM-ToolSandBox: A Unified Framework for Evaluating Visual Tool-Calling Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-07-13",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "MM-ToolSandBox",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "ground progressively arriving visual inputs into correct executable tool calls across a multi-image, multi-turn interaction; handle realistic conversational phenomena mid-task: goal revisions, error corrections, state mutations; operate correctly across 500+ tools spanning 16 application domains; succeed on each of 258 human-verified nominal scenarios (plus 50 interactive-UI variants)",
  "goal_origin": "mixed:given-up-front-with-goal-revisions-injected-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Because visual inputs arrive progressively and goals can be revised mid-task, later tool calls depend on correctly grounding newly arrived visual information and on correctly updating from earlier goal revisions/error corrections/state mutations, so misreading an earlier turn's state corrupts subsequent tool-call decisions.",
  "n_goals": "258 human-verified nominal scenarios plus 50 variants targeting interactive UI applications; 500+ tools across 16 application domains",
  "tracking_demand": "The agent must track a stateful execution environment across multi-image, multi-turn interactions, correctly incorporating goal revisions, error corrections, and state mutations as they occur, and ground each newly arriving visual input into the correct tool call.",
  "scoring": "other:success-rate-plus-failure-mode-decomposition-visual-vs-planning",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the best of 12 evaluated models (from 4B open-weight to frontier proprietary) achieves below 50% success rate; 53% of failures stem from incorrect visual information extraction rather than planning errors. No explicit human/expert baseline is given.",
  "availability": "https://github.com/apple/ml-mmtoolsandbox",
  "goal_span": "Evaluating 12 state-of-the-art models, from 4B open-weight to frontier proprietary systems, shows that current models still lack robust visual tool-calling capability: even the best model achieves below 50% success rate. Our failure analysis further reveals that visual precision, not only planning, is a primary bottleneck for capable models: 53% of failures stem from incorrect information extraction from images despite otherwise correct task workflows.",
  "horizon_span": "supporting multi-image, multi-turn tasks where agents must ground progressively arriving visual inputs into executable tool calls",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290143076",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290990371",
  "title": "OmnilingualGAIA2: Evaluating the Multilingual Gap in Frontier AI Agents",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-08-09",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "OmnilingualGAIA2",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "plan a sequence of tool calls to answer a task; search for information via tools; execute multi-tool workflows; recover from errors during multi-tool execution; all under machine-translated, human-calibrated task instructions across ten languages/five scripts",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tool calls within a task are chained (planning informs search, search results feed execution, execution failures require recovery), so later steps depend on the correctness of earlier tool outputs.",
  "n_goals": null,
  "tracking_demand": "Agent must track its tool-call plan, intermediate search/tool results, and error states to recover from execution failures, now while also handling machine-translated instructions across ten languages including non-Latin scripts.",
  "scoring": "other:pass-at-3 - agents scored via pass@3 across scenario-language pairs; abstract does not describe subgoal-level partial credit within a task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "A universal cross-lingual gap of 8.8-18.4 pass@3 points across seven frontier/open-weight agents, ~55% model-driven vs. only 6.4% attributable to translation contamination; no separate human/expert baseline given.",
  "availability": null,
  "goal_span": "Agentic benchmarks aim to measure how well AI agents plan, search, execute, and recover within realistic multi-tool environments, but they are almost exclusively in English.",
  "horizon_span": "Evaluating seven frontier and open-weight agents, we find a universal cross-lingual gap of 8.8-18.4 pass@3 points that is agent-asymmetric in magnitude, concentrates on tool-orchestration rather than quantitative reasoning, and does not close with model scale.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290990371",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287635340",
  "title": "SeekerGym: A Benchmark for Reliable Information Seeking",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 0,
  "publication_date": "2026-04-18",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "SeekerGym",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "issue repeated retrieval queries to recover as much of a target document's content as possible; quantify uncertainty about how much relevant information might still be missing from what has been retrieved",
  "goal_origin": "given-up-front",
  "decomposition": "open-ended",
  "interdependence": "Successive queries should avoid redundantly re-retrieving already-found passages and instead target still-missing sections, so the agent must track what it has already retrieved to make each subsequent query non-redundant.",
  "n_goals": null,
  "tracking_demand": "The agent must track which passages/sections of the target document it has already retrieved, so it can judge how complete its coverage is and estimate how much information might still be missing.",
  "scoring": "continuous-reward. The abstract reports completeness as a percentage of passages retrieved (42.5% Wikipedia, 29.2% ML Surveys) plus a separate assessment of the agent's uncertainty quantification, not a binary or milestone-based schema.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Best approaches retrieve 42.5% of passages on Wikipedia and 29.2% on ML Surveys; no human/expert comparison given in the abstract.",
  "availability": null,
  "goal_span": "each task in SeekerGym is a document (e.g., a Wikipedia article), and the AI agent must issue queries to retrieve passages from that document... the best approaches retrieve 42.5% of passages on Wikipedia and 29.2% on ML Surveys, leaving substantial room for improvement.",
  "horizon_span": "each episode runs for at most M steps (finite horizon) ... under a fixed budget of M steps with K queries each",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287635340",
  "provenance": "asta-find",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "open-ended",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286222358",
  "title": "SkillCraft: Can LLM Agents Learn to Use Tools Skillfully?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 29,
  "publication_date": "2026-02-28",
  "months_since_pub": 7,
  "citations_per_month": 4.14,
  "artifact_name": "SkillCraft",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "compose atomic tools into reusable higher-level 'Skills'; cache and reuse learned Skills both within a task and across different tasks; complete compositional tool-use scenarios whose difficulty scales with entity count and subtask complexity",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Skills are built by composing atomic tools into reusable higher-level abstractions that are then cached and reused across tasks, so an agent's ability to complete a harder task (more entities, more subtask complexity) depends on the Skills it has already formed and reused from earlier, simpler subtasks/tasks.",
  "n_goals": "126 tasks across 6 difficulty levels, constructed from 21 seed tasks; task-level tool-call counts range from 9 (Easy) to 25 (Hard)",
  "tracking_demand": "The agent must track which higher-level Skills it has already formed and cached, so it can reuse them instead of recomposing atomic tools from scratch on later subtasks/tasks, both within a single task and across the benchmark's 126 tasks.",
  "scoring": "other:efficiency-plus-success-rate. The paper reports 'substantial efficiency gains, with token usage reduced by up to 80% by skill saving and reuse' and that 'success rate strongly correlates with tool composition ability,' jointly crediting task success and skill-reuse efficiency, though no formal subgoal-checkpoint rubric is described.",
  "horizon_value": "126 tasks across 6 difficulty levels (from 21 seed tasks); tool-call counts scale from 9 (Easy: 3 subtasks x 3 calls) to 25 (Hard: 5 subtasks x 5 calls)",
  "horizon_unit": "tool-calls",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (varies by difficulty level)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "SkillCraft features realistic, highly compositional tool-use scenarios with difficulty scaled along both quantitative and structural dimensions, designed to elicit skill abstraction and cross-task reuse. We further propose a lightweight evaluation protocol that enables agents to auto-compose atomic tools into executable Skills, cache and reuse them inside and across tasks.",
  "horizon_span": "3 subtasks x3 API calls = 9 total calls ... 4x4=16 ... 5x5=25 ... 126 tasks across 6 difficulty levels",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286222358",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "tool-calls",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287425055",
  "title": "The Amazing Agent Race: Strong Tool Users, Weak Navigators",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-04-11",
  "months_since_pub": 5,
  "citations_per_month": 0.2,
  "artifact_name": "The Amazing Agent Race (AAR)",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "navigate Wikipedia pages to locate the required entity/fact for each DAG node; execute the correct multi-step tool chain across fork-merge branches linking extracted entities; aggregate branch outputs into one verifiable final answer",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Fork-merge DAG structure means downstream 'merge' nodes depend on multiple parallel upstream branches all completing correctly before their outputs can be aggregated, unlike a simple linear chain.",
  "n_goals": "1,400 instances: 800 sequential legs and 600 compositional DAG legs; DAG legs average 22 pit-stops (nodes) with up to 5 diamonds (fork-merge structures)",
  "tracking_demand": "The agent must track which Wikipedia entities/facts it has already extracted at each DAG node, correctly route them through parallel fork-merge tool-chain branches, and retain intermediate values needed for final aggregation across up to 5 diamonds per leg.",
  "scoring": "subgoal-checkpoint-partial-credit -- three complementary metrics (finish-line accuracy, pit-stop visit rate, roadblock completion rate) separately credit navigation, tool-use, and arithmetic sub-steps rather than only a single final-answer score.",
  "horizon_value": "average of 22.1 pit stops per leg for AAR-DAG (600 legs) and 15.0 pit stops per leg for AAR-Linear (800 legs), up to 5 diamonds per leg",
  "horizon_unit": "other:pit-stops (navigation hops)",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per leg/instance)",
  "horizon_stated": "yes",
  "headline_result": "Best agent framework achieves only 37.2% accuracy (Claude Code matches Codex CLI at ~37% with 6x fewer tokens); no human/expert baseline is reported in the abstract.",
  "availability": "https://minnesotanlp.github.io/the-amazing-agent-race",
  "goal_span": "Three complementary metrics (finish-line accuracy, pit-stop visit rate, and roadblock completion rate) separately diagnose navigation, tool-use, and arithmetic failures.",
  "horizon_span": "with an average of 22 pit stops and up to 5 diamonds",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287425055",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290093158",
  "title": "ToolGym: an Open-world Tool-using Environment for Scalable Agent Testing and Data Curation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 5,
  "publication_date": null,
  "months_since_pub": 3,
  "citations_per_month": 1.67,
  "artifact_name": "ToolGym",
  "artifact_kind": "environment/simulator",
  "domain": "tool-use-API",
  "goal_types": "complete long-horizon, multi-tool workflows synthesized with wild constraints across 5,571 tools and 204 apps; recover and adapt when a state controller injects interruptions and failures mid-workflow; separate deliberate planning/self-correction from step-wise execution via a planner-actor decomposition",
  "goal_origin": "mixed:given-up-front-with-injected-interruptions-and-failures",
  "decomposition": "sequential-chain",
  "interdependence": "A state controller injects interruptions and failures mid-workflow, so completing later tool calls depends on the planner/actor correctly re-tracking goals and recovering from disruptions introduced earlier in the same long-horizon workflow.",
  "n_goals": "5,571 format-unified tools across 204 commonly used apps; 1,170 collected trajectories used for fine-tuning",
  "tracking_demand": "Agent must track tool-call state and wild constraints across a long-horizon, multi-tool workflow while detecting and recovering from injected interruptions/failures and unreliable tool states.",
  "scoring": "other:misalignment-analysis-between-planning-and-execution",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "It includes a task creation engine that synthesizes long-horizon, multi-tool workflows with wild constraints, and a state controller that injects interruptions and failures to stress-test robustness. On top of this environment, we develop a tool select-then-execute agent framework with a planner-actor decomposition to separate deliberate reasoning and self-correction from step-wise execution.",
  "horizon_span": "It includes a task creation engine that synthesizes long-horizon, multi-tool workflows with wild constraints",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290093158",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290331838",
  "title": "ToolVerse: Unlocking Massive Environments and Long-Horizon Tasks for Agentic Reinforcement Learning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 2,
  "publication_date": "2026-07-17",
  "months_since_pub": 2,
  "citations_per_month": 1.0,
  "artifact_name": "ToolVerse / GUST dataset",
  "artifact_kind": "dataset",
  "domain": "tool-use-API",
  "goal_types": "complete a long-horizon task built from a tool-dependency graph, where later tools unlock only after prerequisite tool-use subgoals are completed; correctly integrate the right subset of tools from a large pool (~4500 tools across ~400 MCPs) into a coherent multi-step solution; receive properly assigned turn-level credit despite long sequences of tool calls",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are generated from a tool-dependency graph via a 'Dynamic Unlocking Sampling Algorithm,' meaning later tool-use subgoals are only reachable/meaningful once earlier, prerequisite tool-use subgoals in the graph have been completed.",
  "n_goals": "tasks built from ~4500 tools across ~400 real-world MCPs, via a tool-dependency graph (GUST dataset)",
  "tracking_demand": "The agent must track which prerequisite tools/subgoals in the dependency graph it has already satisfied in order to know which further tools are unlocked and relevant next, across a long sequence of tool calls.",
  "scoring": "other:turn-aware-credit-assignment. The paper proposes a 'fine-grained Turn-Aware Relative Advantage algorithm' specifically to 'alleviate the credit assignment problem in long-horizon agentic RL,' implying turn/subgoal-level credit during training, though the abstract does not describe a formal held-out evaluation rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we propose a task design strategy based on a tool dependency graph, utilizing Dynamic Unlocking Sampling Algorithm to generate long-horizon tasks, and produce GUST (Graph Unlocking Sampling Tasks) dataset.",
  "horizon_span": "tasks consist of 3 to 7 tasks per data item",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290331838",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285974241",
  "title": "Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 3,
  "publication_date": "2026-02-22",
  "months_since_pub": 7,
  "citations_per_month": 0.43,
  "artifact_name": "Toolathlon",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "complete each of 108 real-world tool-use tasks by following a canonical multi-step tool-invocation solution path; stay within the operating envelope of the canonical path across the whole trajectory to avoid stochastic drift/derailment",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each tool call's correctness depends on adherence to the prior sequence of calls: once a trajectory drifts off the canonical path, each subsequent off-canonical call raises the probability that the next call is also off-canonical, a self-reinforcing failure cascade (+22.7 percentage points per off-canonical step).",
  "n_goals": "108 real-world tool-use tasks; 22 models x 108 tasks x 3 runs = 515 model-by-task units analyzed",
  "tracking_demand": "Success requires the trajectory of tool calls to stay within the operating envelope of the task's canonical solution path; mid-trajectory adherence must be tracked since drift compounds over subsequent calls.",
  "scoring": "other:trajectory-canonical-path-adherence-plus-task-success. The study measures both binary task success/failure (from Toolathlon) and a continuous Jaccard-similarity canonical-path-adherence metric per trajectory, i.e., a process-level signal in addition to final outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "A monitor that restarts the bottom tercile of runs based on mid-trajectory canonical adherence lifts success rates by +8.8 percentage points among intervened runs; no human/expert baseline given.",
  "availability": null,
  "goal_span": "We analyze trajectories from the Toolathlon benchmark: 22 frontier models each attempt 108 real-world tool-use tasks across 3 independent runs, yielding 515 model$\\times$task units where the same model succeeds on some runs and fails on others due to LLM sampling stochasticity alone.",
  "horizon_span": "108 real-world tool-use tasks across 3 independent runs, yielding 515 model$\\times$task units",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285974241",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290014383",
  "title": "UniClawBench: A Universal Benchmark for Proactive Agents on Real-World Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2026-07-09",
  "months_since_pub": 2,
  "citations_per_month": 0.5,
  "artifact_name": "UniClawBench",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "demonstrate correct skill usage for the task's required tool/capability; explore the live environment (e.g. a Docker container) to discover needed information/actions; reason correctly over long context accumulated during the task; correctly interpret multimodal inputs relevant to the task; coordinate actions correctly across multiple platforms/services",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "A task's step-by-step completion checkpoints must generally be satisfied in sequence within the live Docker environment, and the closed-loop design (hidden supervisor agent, user simulator agent) means later steps depend on multi-turn feedback exchanged earlier in the interaction without leaking grading criteria.",
  "n_goals": "400 bilingual real-world tasks organized around five foundational model capabilities (Skill Usage, Exploration, Long-Context Reasoning, Multimodal Understanding, Cross-Platform Coordination)",
  "tracking_demand": "The agent must track its progress against fine-grained, step-by-step completion checkpoints within a live environment, integrate multi-turn feedback from a hidden supervisor agent and a user-simulator agent without seeing the grading criteria, and coordinate across the relevant capability dimensions needed for that task.",
  "scoring": "subgoal-checkpoint-partial-credit -- the benchmark evaluates agents in live Docker containers using fine-grained, step-by-step completion checkpoints, explicit non-binary, per-step credit rather than a single pass/fail per task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/HKU-MMLab/UniClawBench",
  "goal_span": "Unlike previous benchmarks that rely on static, pre-recorded answers, our benchmark evaluates agents in live Docker containers using fine-grained, step-by-step completion checkpoints. Furthermore, we design a closed-loop evaluation strategy comprising an executor agent, a hidden supervisor agent, and a user agent to simulate realistic multi-turn human feedback without leaking grading criteria.",
  "horizon_span": "our benchmark evaluates agents in live Docker containers using fine-grained, step-by-step completion checkpoints... we design 400 bilingual real-world tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290014383",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279118509",
  "title": "CONFETTI: Conversational Function-Calling Evaluation Through Turn-Level Interactions",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 16,
  "publication_date": "2025-06-02",
  "months_since_pub": 15,
  "citations_per_month": 1.07,
  "artifact_name": "CONFETTI",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "handle user follow-up requests within an ongoing conversation; correct or switch goals mid-conversation when the user changes intent; resolve ambiguous or implicit user goals into an appropriate API call; chain multiple function calls together to satisfy a multi-step user request",
  "goal_origin": "mixed:given-up-front-with-user-injected-goal-changes-mid-conversation",
  "decomposition": "sequential-chain",
  "interdependence": "Later turns must correctly interpret goal corrections/switches and ambiguous references established earlier, and chained function calls depend on the outputs of prior calls.",
  "n_goals": null,
  "tracking_demand": "The agent must track evolving user intent across conversation turns, recognize goal correction/switching, and maintain state of prior chained function-call results across up to 86 available APIs.",
  "scoring": "subgoal-checkpoint-partial-credit. Evaluation is explicitly turn-level (off-policy per-turn function-calling accuracy plus dialog-act annotations), i.e., graded at each turn rather than only on whole-conversation success.",
  "horizon_value": "313 user turns across 109 conversations (~2.9 turns/conversation on average)",
  "horizon_unit": "turns",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per whole episode (per conversation); the paper reports totals (313 turns, 109 conversations) rather than stating a per-conversation average directly.",
  "horizon_stated": "yes",
  "headline_result": "Top models: Nova Pro (40.01%), Claude Sonnet v3.5 (35.46%), Llama 3.1 405B (33.19%); no human/expert baseline stated.",
  "availability": null,
  "goal_span": "These conversations explicitly target various conversational complexities, such as follow-ups, goal correction and switching, ambiguous and implicit goals. We perform off-policy turn-level evaluation using this benchmark targeting function-calling.",
  "horizon_span": "CONFETTI addresses this gap through 109 human-simulated conversations1, comprising 313 user turns and covering 86 APIs.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279118509",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278769440",
  "title": "Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "web-GUI",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 17,
  "publication_date": "2025-05-19",
  "months_since_pub": 16,
  "citations_per_month": 1.06,
  "artifact_name": "DialogTool / VirtualMobile",
  "artifact_kind": "dataset",
  "domain": "tool-use-API",
  "goal_types": "create a new tool/API on demand within a multi-turn dialogue (tool creation); become aware of, correctly select, and execute the right tool given user intent (tool utilization); generate a role-consistent response, including role play, reflecting whether/how the tool was used",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "The whole life cycle of tool use is staged: tool creation must precede tool utilization (awareness, selection, execution) for a newly created tool, and both precede generating a role-consistent response, so state built in earlier turns constrains later turns of the same multi-turn dialogue.",
  "n_goals": "six key tasks across three stages (tool creation; tool awareness, selection, execution; response generation and role play)",
  "tracking_demand": "Agent must track which tools have been created/exist, their stateful execution history across the dialogue, and how to remain role-consistent while using them, across a multi-turn conversation.",
  "scoring": "other:per-stage-evaluation-across-13-llms - comprehensive evaluation on 13 distinct open- and closed-source LLMs with detailed per-stage analysis (tool creation, awareness, selection, execution, response generation, role play); implies per-stage (subgoal-level) reporting rather than one aggregate pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we propose \\texttt{DialogTool}, a multi-turn dialogue dataset with stateful tool interactions considering the whole life cycle of tool use, across six key tasks in three stages: 1) \\textit{tool creation}; 2) \\textit{tool utilization}: tool awareness, tool selection, tool execution; and 3) \\textit{role-consistent response}: response generation and role play.",
  "horizon_span": "revealing that the existing state-of-the-art LLMs still cannot perform well to use tools over long horizons.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278769440",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283243734",
  "title": "M^3-Bench: Multi-Modal, Multi-Hop, Multi-Threaded Tool-Using MLLM Agent Benchmark",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 9,
  "publication_date": "2025-11-21",
  "months_since_pub": 10,
  "citations_per_month": 0.9,
  "artifact_name": "M^3-Bench",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "complete multi-hop, multi-threaded tool-call workflows requiring cross-tool dependencies; maintain persistence of intermediate resources across steps; ground visual and textual reasoning correctly across each tool call",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Cross-tool dependencies mean later tool calls require intermediate resources or results produced by earlier calls in the same or a parallel thread, and multi-threaded workflows interleave chains that must still be correctly aligned.",
  "n_goals": "28 servers with 231 tools; per-task subgoal (tool-call) count not stated in the abstract",
  "tracking_demand": "Agent must track intermediate resources produced by earlier tool calls and align them across multiple concurrent threads for argument fidelity and structural consistency.",
  "scoring": "other:similarity-bucketed-hungarian-matching-plus-llm-ensemble - reports interpretable metrics decoupling semantic fidelity from workflow consistency, plus a four-LLM judge ensemble reporting Task Completion and information grounding; the auditable one-to-one tool-call matching implies per-call (subgoal-level) credit rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/EtaYang10th/Open-M3-Bench",
  "goal_span": "The benchmark targets realistic, multi-hop and multi-threaded workflows that require visual grounding and textual reasoning, cross-tool dependencies, and persistence of intermediate resources across steps.",
  "horizon_span": "The benchmark spans 28 servers with 231 tools, and provides standardized trajectories curated through an Executor&Judge pipeline with human verification.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283243734",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "277509805",
  "title": "Multi-Mission Tool Bench: Assessing the Robustness of LLM based Agents through Related and Dynamic Missions",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 7,
  "publication_date": "2025-04-03",
  "months_since_pub": 17,
  "citations_per_month": 0.41,
  "artifact_name": "Multi-Mission Tool Bench",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "complete each mission within a test case containing multiple interrelated missions; dynamically adapt when missions switch mid-interaction; correctly invoke tools appropriate to the currently active mission; handle all possible mission-switching patterns within a fixed mission number",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Multiple interrelated missions within one test case share context and tool state; a mission switch requires correctly recognizing the new active mission without losing track of paused missions, and the benchmark explores all possible mission-switching patterns within a fixed mission number, meaning a given case may require handling several switches in sequence.",
  "n_goals": "multiple interrelated missions per test case; all possible mission-switching patterns explored within a fixed mission number (exact count not stated in the abstract)",
  "tracking_demand": "The agent must track the state of each interrelated mission (active or paused), correctly recognize mission switches, and select/invoke the correct tools for whichever mission is currently active, evaluated via dynamic decision trees for accuracy and efficiency.",
  "scoring": "other:decision-tree-based-accuracy-and-efficiency. A novel method evaluates the accuracy and efficiency of agent decisions using dynamic decision trees, a structured, process-aware (non-binary) evaluation of the sequence of mission-switching decisions rather than a single final success/failure label.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "In the benchmark, each test case comprises multiple interrelated missions. This design requires agents to dynamically adapt to evolving demands. Moreover, the proposed benchmark explores all possible mission-switching patterns within a fixed mission number.",
  "horizon_span": "the proposed benchmark explores all possible mission-switching patterns within a fixed mission number",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277509805",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282401329",
  "title": "OrchDAG: Complex Tool Orchestration in Multi-Turn Interactions with Plan DAGs",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 1,
  "publication_date": "2025-10-28",
  "months_since_pub": 11,
  "citations_per_month": 0.09,
  "artifact_name": "OrchDAG",
  "artifact_kind": "dataset",
  "domain": "tool-use-API",
  "goal_types": "correctly execute a sequence of tool calls whose dependencies form a directed acyclic graph (DAG); respect topological/precedence constraints among tool calls across multi-turn interactions; solve DAG-structured tool-orchestration tasks of controllable/varying complexity",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tool calls are explicitly generated with DAG dependency structure, so a later tool call may require outputs or preconditions established by one or more earlier tool calls in the graph, and correctness depends on respecting this topological order across multiple turns.",
  "n_goals": null,
  "tracking_demand": "The agent must track which nodes (tool calls) in the DAG have been completed and what state/outputs they produced, to correctly select and sequence subsequent tool calls consistent with the graph's precedence constraints across multiple turns.",
  "scoring": "other:benchmark-score-plus-graph-based-RLVR-reward",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We introduce OrchDAG, a synthetic data generation pipeline that models tool execution as directed acyclic graphs (DAGs) with controllable complexity. Using this dataset, we benchmark model performance and propose a graph-based reward to enhance RLVR training.",
  "horizon_span": "yet most existing work overlooks the complexity of multi-turn tool interactions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282401329",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282574909",
  "title": "The Tool Decathlon: Benchmarking Language Agents for Diverse, Realistic, and Long-Horizon Task Execution",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 66,
  "publication_date": "2025-10-29",
  "months_since_pub": 11,
  "citations_per_month": 6.0,
  "artifact_name": "Tool Decathlon (Toolathlon)",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "coordinate interactions across multiple named Apps/tools (e.g. email + calendar + file systems) to complete one complex workflow; diagnose and report anomalies via monitoring a database following an operating manual; satisfy a strictly execution-verifiable end state for each of 108 tasks spanning 32 apps and 604 tools",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Completing a task typically requires coordinating across multiple Apps whose states are interdependent, and the realistic initial environment states mean each tool call must remain consistent with the current state of other apps.",
  "n_goals": "108 tasks across 32 apps and 604 tools; tasks require interacting with multiple Apps over around 20 turns on average",
  "tracking_demand": "The agent must track the current, realistic state of multiple named applications across roughly 20 tool-calling turns per task, and verify its final state against a dedicated evaluation script.",
  "scoring": "binary-final-success -- each task is strictly verifiable through dedicated evaluation scripts; the abstract does not describe finer intra-task subgoal-checkpoint partial credit beyond this final verification, though it reports average tool-calling turns as a secondary efficiency metric.",
  "horizon_value": "tasks require interacting with multiple Apps over around 20 turns on average (best model averages 20.2 tool-calling turns)",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Best-performing model, Claude-4.5-Sonnet, achieves only a 38.6% success rate with 20.2 tool-calling turns on average, while the top open-weights model DeepSeek-V3.2-Exp reaches 20.1%; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Toolathlon spans 32 software applications and 604 tools, ranging from everyday platforms such as Google Calendar and Notion to professional ones like WooCommerce, Kubernetes, and BigQuery... This benchmark includes 108 manually sourced or crafted tasks in total, requiring interacting with multiple Apps over around 20 turns on average to complete. Each task is strictly verifiable through dedicated evaluation scripts.",
  "horizon_span": "This benchmark includes 108 manually sourced or crafted tasks in total, requiring interacting with multiple Apps over around 20 turns on average to complete... the best-performing model, Claude-4.5-Sonnet, achieves only a 38.6% success rate with 20.2 tool calling turns on average.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282574909",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "278995670",
  "title": "ToolHaystack: Stress-Testing Tool-Augmented Language Models in Realistic Long-Term Interactions",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "tool-use-API",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "tool-use-API",
  "citation_count": 5,
  "publication_date": "2025-05-29",
  "months_since_pub": 16,
  "citations_per_month": 0.31,
  "artifact_name": "ToolHaystack",
  "artifact_kind": "benchmark",
  "domain": "tool-use-API",
  "goal_types": "correctly maintain and disambiguate multiple concurrent task-execution contexts within one continuous conversation; handle realistic noise/disruptions injected into a long-term interaction without losing track of tool-use context; successfully complete tool-use tasks embedded within a long, continuous conversation despite these challenges",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "Multiple task-execution contexts are interleaved within a single continuous conversation alongside realistic noise, so correctly handling one embedded task requires the model to keep it separated from other concurrent contexts and disruptions rather than conflating them.",
  "n_goals": null,
  "tracking_demand": "The model must maintain and disambiguate multiple concurrent task-execution contexts across a continuous long-term conversation while filtering out realistic noise and handling various disruptions, rather than resetting context between short, isolated tool-use exchanges.",
  "scoring": "other:performance-comparison-standard-multi-turn-vs-ToolHaystack-long-term",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Current models perform well in standard multi-turn tool-use settings but struggle significantly on ToolHaystack, revealing critical long-term robustness gaps not seen in prior tool benchmarks; no specific numeric top score or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "By applying this benchmark to 14 state-of-the-art LLMs, we find that while current models perform well in standard multi-turn settings, they often significantly struggle in ToolHaystack, highlighting critical gaps in their long-term robustness not revealed by previous tool benchmarks.",
  "horizon_span": "offering limited insight into model behavior during realistic long-term interactions",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278995670",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "291382713",
  "title": "Behavior2Trip: Towards Personalized Travel Planning via User Behavior Trajectory",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 0,
  "publication_date": "2026-08-27",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "Behavior2Trip",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "infer a user's latent travel preferences from their past behavior trajectory rather than explicit instructions; generate a travel plan satisfying preferences across 14 attributes spanning 5 preference dimensions; achieve a full-constraint pass across all inferred preference constraints simultaneously",
  "goal_origin": "implied-by-constraints",
  "decomposition": "set-of-independent",
  "interdependence": "Preference constraints across the 5 dimensions and 14 attributes are inferred from one shared behavior history and must all be jointly satisfied in a single generated plan, so satisfying one preference (e.g., budget) can conflict with another (e.g., a preferred activity type).",
  "n_goals": "14 attributes across 5 preference dimensions per instance, inferred from an average of 39.8 past user behaviors",
  "tracking_demand": "The agent must infer and track the user's latent preferences (14 attributes, 5 dimensions) from an average of 39.8 past behaviors, then check the generated plan against every inferred constraint for a full pass.",
  "scoring": "other:full-constraint-pass-rate. The paper explicitly reports a 'full-constraint pass rate' metric (GPT-4.1 scores only 0.5% on the hardest tasks), a strict all-constraints-satisfied criterion rather than itemized per-constraint partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "GPT-4.1 achieves a full-constraint pass rate of only 0.5% on the hardest tasks; B2T-Agent (Qwen3-8B) outperforms all baselines and also outperforms GPT-4.1 on the separate TravelPlanner benchmark; no human/expert baseline given.",
  "availability": "https://github.com/BUAA-IRIP-LLM/Behavior2Trip",
  "goal_span": "Each instance represents an average of 39.8 past user behaviors spanning 14 attributes across 5 preference dimensions.",
  "horizon_span": "Each instance represents an average of 39.8 past user behaviors spanning 14 attributes across 5 preference dimensions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291382713",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285050875",
  "title": "DeepPlanning: Benchmarking Long-Horizon Agentic Planning with Verifiable Constraints",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 33,
  "publication_date": "2026-01-26",
  "months_since_pub": 8,
  "citations_per_month": 4.12,
  "artifact_name": "DeepPlanning",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "satisfy local, fine-grained constraints on individual itinerary/shopping items; satisfy global constrained optimization objectives (e.g. time and financial budgets) across the whole plan; proactively gather information needed before constraints can even be checked",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Global budget/time constraints span the whole multi-day itinerary or multi-product shopping list, so a locally-optimal choice for one day/item can violate the global optimization target, requiring the agent to trade off local reasoning against global constraints while still gathering missing information.",
  "n_goals": null,
  "tracking_demand": "Agent must track accumulated time and financial budget consumption across a multi-day plan or multi-product list, alongside fine-grained local constraints, while still actively gathering new information mid-task.",
  "scoring": "other:not-stated \u2014 the abstract states frontier agentic LLMs 'struggle with these problems' but does not describe a specific partial-credit/checkpoint scoring scheme.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even frontier agentic LLMs struggle with DeepPlanning tasks (no specific numeric score or human baseline given in the abstract).",
  "availability": "code and data open-sourced (per abstract: 'We open-source the code and data'); no specific URL given.",
  "goal_span": "It features multi-day travel planning and multi-product shopping tasks that require proactive information acquisition, local constrained reasoning, and global constrained optimization.",
  "horizon_span": "It features multi-day travel planning and multi-product shopping tasks",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285050875",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "288669343",
  "title": "GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 4,
  "publication_date": "2026-05-24",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "GroupTravelBench",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "elicit each group member's private travel preferences through multi-turn dialogue; surface and resolve inter-user preference conflicts via compromise or subgrouping; produce a final plan that balances group utility against fairness across all members",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "DAG-with-precedence",
  "interdependence": "A valid final plan depends on having elicited each member's private preferences first, then resolved conflicts among them, so later planning steps are gated on the elicitation and coordination sub-goals having been satisfied for all group members.",
  "n_goals": "650 tasks across three difficulty levels; N group members per task with individually elicited preferences",
  "tracking_demand": "Agent must track each group member's elicited (and initially private) preferences, detected conflicts between members, and the evolving fairness/utility trade-off of the plan across a synchronous multi-turn group-chat session.",
  "scoring": "other:mixed \u2014 'a complementary evaluation framework combining rule-based outcome metrics and LLM-judge process metrics'; explicit rule-based outcome metrics (e.g. plan validity below 12%) imply checkpoint-style outcome scoring alongside LLM-judge process-level assessment.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the strongest agents fall short on all four rule-based outcome metrics, with plan validity below 12% (no human baseline given).",
  "availability": null,
  "goal_span": "GroupTravelBench probes three group-specific capabilities: \\textit{(i) elicitation} of private preferences through multi-turn dialogue; \\textit{(ii) coordination} of inter-user conflicts via compromise or subgrouping; and \\textit{(iii) planning} that balances group utility against fairness",
  "horizon_span": "it comprises 650 tasks across three difficulty levels, each running in a synchronous group-chat sandbox with cached tool data for reproducible offline evaluation",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288669343",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290648300",
  "title": "TREK: A Travel Reasoning and Evaluation Kit for LLM Agents in Complex Trip Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 0,
  "publication_date": "2026-07-29",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "TREK",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "produce a single itinerary that is jointly constraint-correct, hallucination-free, spatio-temporally executable, and budget-valid; respond to the traveler's unstated persona needs; correctly identify provably infeasible tasks (267 of 800) versus feasible ones with typed causes",
  "goal_origin": "given-up-front",
  "decomposition": "other:joint-constraint-satisfaction-within-one-itinerary",
  "interdependence": "Every flight, hotel, and attraction choice must be simultaneously bookable, physically traversable across days, within budget, and responsive to unstated persona needs, so satisfying one constraint (e.g., budget) can break another (e.g., spatio-temporal feasibility).",
  "n_goals": "800 multi-constraint tasks (533 feasible, 267 provably infeasible) evaluated across nine constraint dimensions",
  "tracking_demand": "Agent must track running budget, spatio-temporal feasibility across days, entity/route validity, and unstated persona needs simultaneously while assembling a single itinerary.",
  "scoring": "other:deterministic-rule-based-evaluator-with-gold-reference",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the strongest model (GPT-5.6) produces a fully-feasible plan on only 46.2% of solvable tasks, with a median of 6.6% and a floor of 0.0%; a human-verified gold reference scores a perfect 1.0, showing satisfying unstated persona needs is the universal, unsolved bottleneck.",
  "availability": null,
  "goal_span": "Every task is scored by a fully deterministic, rule-based evaluator with no LLM judge and ships a human-verified gold reference that scores a perfect 1.0 under that same evaluator... Evaluating 15 LLM agents across nine constraint dimensions, we find that even the strongest (GPT-5.6) produces a fully-feasible plan on only 46.2% of solvable tasks",
  "horizon_span": "TREK comprises 800 multi-constraint tasks - 533 feasible and 267 provably infeasible with typed route/entity/budget causes - over a synthetic, internally consistent knowledge base of 212,530 records across 375 cities and 13 personas",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290648300",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "programmatic verifier",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285269849",
  "title": "TRIP-Bench: A Benchmark for Long-Horizon Interactive Agents in Real-World Scenarios",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 7,
  "publication_date": "2026-02-02",
  "months_since_pub": 7,
  "citations_per_month": 1.0,
  "artifact_name": "TRIP-Bench",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "satisfy each of 40+ curated travel requirements/global constraints across an itinerary; coordinate reasoning across 18 curated tools correctly within one dialogue; adapt to evolving user behavior, style shifts, feasibility changes, and iterative version revisions over a long, multi-turn interaction (hard split)",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Global travel constraints and previously committed itinerary choices must remain jointly satisfied as new user turns introduce style shifts, feasibility changes, or revision requests, so later dialogue turns must reconcile new information with all 40+ requirements and prior tool-call results.",
  "n_goals": "18 curated tools and 40+ travel requirements; dialogues span up to 15 user turns, can involve 150+ tool calls, and may exceed 200k tokens of context",
  "tracking_demand": "The agent must track global constraint satisfaction across 40+ travel requirements, the state of 18 tools' results, and evolving user preferences/style/feasibility across dialogues spanning up to 15 user turns, 150+ tool calls, and 200k+ tokens of context.",
  "scoring": "continuous-reward -- performance is measured via constraint-satisfaction success rate, with even advanced models achieving at most 50% success on the easy split and below 10% on hard subsets; the abstract does not describe a discrete subgoal-checkpoint partial-credit rubric distinct from overall constraint satisfaction.",
  "horizon_value": "dialogues span up to 15 user turns, can involve 150+ tool calls, and may exceed 200k tokens of context",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task/dialogue",
  "horizon_stated": "yes",
  "headline_result": "Even advanced models achieve at most 50% success on the easy split, dropping below 10% on hard subsets; the paper's own GTPO method (applied to Qwen2.5-32B-Instruct) outperforms Gemini-3-Pro in this evaluation. No human/expert baseline is reported.",
  "availability": null,
  "goal_span": "TRIP-Bench leverages real-world data, offers 18 curated tools and 40+ travel requirements, and supports automated evaluation. It includes splits of varying difficulty; the hard split emphasizes long and ambiguous interactions, style shifts, feasibility changes, and iterative version revision. Dialogues span up to 15 user turns, can involve 150+ tool calls, and may exceed 200k tokens of context.",
  "horizon_span": "Dialogues span up to 15 user turns, can involve 150+ tool calls, and may exceed 200k tokens of context.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285269849",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288862172",
  "title": "TravelEval: A Comprehensive Benchmarking Framework for Evaluating LLM-Powered Travel Planning Agents",
  "year": 2026,
  "venue": "Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 1,
  "publication_date": "2026-05-31",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "TravelEval",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "produce a full multi-day itinerary satisfying accuracy, compliance, temporality, spatiality, economy, and utility dimensions jointly; sequence daily accommodation, transport, and visit pacing consistently across the whole trip rather than per isolated day",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Daily plans share accommodation continuity, inter-city transport links, and cumulative budget/time constraints, so an earlier day's choices constrain what is feasible on later days.",
  "n_goals": "itineraries span 2 to 7 days (six duration categories) with six evaluation dimensions applied to the whole plan",
  "tracking_demand": "The agent must track cumulative spatio-temporal cost (queuing times, transit distances), running budget/economy, and daily pacing/accommodation continuity across the entire multi-day itinerary, not just within a single day.",
  "scoring": "milestone-rubric -- a six-dimensional evaluation framework (accuracy, compliance, temporality, spatiality, economy, utility) scores the whole plan; the abstract does not describe finer per-goal partial credit beyond the six dimensions.",
  "horizon_value": "itineraries span 2-day, 3-day, 4-day, 5-day, 6-day, and 7-day durations across query categories (e.g. 400 medium-difficulty queries)",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per itinerary/trip)",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/onlycwy11/TravelEval",
  "goal_span": "TravelEval features 1) a novel six-dimensional evaluation framework to holistically assess plans across accuracy, compliance, temporality, spatiality, economy, and utility dimensions; 2) a highly realistic data sandbox with precise accommodation pricing and authentic intercity transportation data; and 3) a simulation-based global evaluation method that emulates complete travel plans with API-integrated geographic information and fine-grained queuing time.",
  "horizon_span": "queries are distributed across 2-day, 3-day, 4-day, 5-day, 6-day, 7-day categories, with the largest group being 400 medium-difficulty queries",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288862172",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289625995",
  "title": "Trip+: Benchmarking Agents in Personalized Interactive Travel Planning",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 1,
  "publication_date": "2026-06-19",
  "months_since_pub": 3,
  "citations_per_month": 0.33,
  "artifact_name": "Trip+",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "generate a minute-level itinerary satisfying a traveler's profiled preferences; revise the itinerary in response to evolving preferences and unexpected environment-driven disruptions across multiple turns; avoid producing technically-feasible-but-exhausting plans (jointly satisfy feasibility and experiential/fatigue quality)",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Each itinerary revision must remain consistent with previously committed plan elements while incorporating new disruptions/preferences, and minute-level daily pacing decisions accumulate into overall traveler fatigue, so early over-packing constrains what remains experientially feasible later.",
  "n_goals": "153 multi-turn instances with 570 user turns total (roughly 3.7 user turns/instance on average)",
  "tracking_demand": "The agent must track the traveler's evolving profile/preferences, the current committed itinerary state at minute-level granularity, and cumulative experiential cost (e.g. fatigue) as it revises plans across several user turns per instance.",
  "scoring": "LLM-judge-rubric -- an LLM-based simulator evaluates end-to-end traveler experience including subjective metrics like fatigue, going beyond feasibility-only checks; the abstract does not describe a fixed points rubric, but this is judge-based, multi-dimensional (not single binary) scoring.",
  "horizon_value": "153 multi-turn instances and 570 user turns total",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "dataset-wide total, implying an average of roughly 3.7 user turns per instance/episode",
  "horizon_stated": "yes",
  "headline_result": "18 LMs show a consistent gap in experiential quality: models favor technically feasible but exhausting itineraries that diverge sharply from profiled traveler preferences; no specific best-model score or human baseline number is given in the abstract.",
  "availability": null,
  "goal_span": "In Trip+, given traveler profiles and dynamic interactions, agents must generate and revise minute-level itineraries. End-to-end traveler experiences are evaluated via an LLM-based simulator, enabling the assessment of subjective metrics like fatigue.",
  "horizon_span": "153 multi-turn instances and 570 user turns",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289625995",
  "provenance": "asta-find",
  "scoring_family": "LLM-judge-rubric",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287929615",
  "title": "Agentic AI for Trip Planning Optimization Application",
  "year": 2026,
  "venue": "2026 IEEE Intelligent Vehicles Symposium (IV)",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 0,
  "publication_date": "2026-04-30",
  "months_since_pub": 5,
  "citations_per_month": 0.0,
  "artifact_name": "Trip-planning Optimization Problems (TOP) Dataset",
  "artifact_kind": "dataset",
  "domain": "travel-and-constraint-planning",
  "goal_types": "optimize (not merely satisfy) route/itinerary selection under travel-time, energy, and traffic factors; coordinate specialized sub-agents for traffic, charging, and points-of-interest; dynamically refine a plan against a definitive optimal solution reference",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "An orchestration agent coordinates specialized traffic, charging, and POI sub-agents whose outputs jointly determine the overall route's optimality; refining one sub-agent's recommendation (e.g. a charging stop) can change what remains optimal for the traffic or POI sub-agents' recommendations.",
  "n_goals": "category-level task structure with fine-grained analysis (exact task count not given in abstract)",
  "tracking_demand": "The orchestration agent must track recommendations from the traffic, charging, and POI sub-agents and reconcile them into one jointly-optimal plan, verified against the dataset's definitive optimal solutions rather than only a feasible reference answer.",
  "scoring": "other:accuracy-against-ground-truth-optimal-solution",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "The system achieves 77.4% accuracy on the TOP Benchmark, significantly outperforming single-agent and workflow-based multi-agent baselines; no human baseline given.",
  "availability": null,
  "goal_span": "we address these limitations with an agentic AI framework that enables dynamic refinement through an orchestration agent coordinating specialized agents for traffic, charging, and points of interest, and with the Trip-planning Optimization Problems Dataset, which supplies definitive optimal solutions and category-level task structure for fine-grained analysis.",
  "horizon_span": "our system achieves 77.4% accuracy on the TOP Benchmark, significantly outperforming single-agent and workflow-based multi-agent baselines",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287929615",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285452575",
  "title": "WorldTravel: A Realistic Multimodal Travel-Planning Benchmark with Tightly Coupled Constraints",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 3,
  "publication_date": "2026-02-09",
  "months_since_pub": 7,
  "citations_per_month": 0.43,
  "artifact_name": "WorldTravel / WorldTravel-Webscape",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "satisfy all of an average 15+ interdependent temporal and logical travel constraints simultaneously per scenario; extract constraint parameters from dynamic web environments/webpages rather than idealized data; perceive constraint parameters directly from visual layouts in a multi-modal setting; produce a feasible travel plan across 150 real-world scenarios in 5 cities",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Constraints are tightly coupled: a single early decision determines the feasibility of all subsequent actions, unlike loosely-coupled benchmarks solvable by local greedy choices, so early errors cascade through the rest of the itinerary.",
  "n_goals": "an average of 15+ interdependent constraints per scenario; 150 scenarios across 5 cities",
  "tracking_demand": "The agent must track which of the 15+ interdependent temporal/logical constraints have been satisfied so far, and in the multi-modal condition must also perceive constraint parameters directly from over 2,000 rendered webpages rather than being handed clean structured data.",
  "scoring": "other:feasibility-rate-text-only-vs-multimodal",
  "horizon_value": "an average of 15+ interdependent constraints per scenario; a Planning Horizon threshold at approximately 10 constraints",
  "horizon_unit": "other:number-of-interdependent-constraints",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (per travel-planning scenario)",
  "horizon_stated": "yes",
  "headline_result": "Even the state-of-the-art GPT-5.2 achieves only 32.67% feasibility in text-only settings, dropping to 19.33% in multi-modal environments; no human/expert baseline is reported, though the authors identify a Planning Horizon threshold at ~10 constraints where reasoning consistently fails.",
  "availability": null,
  "goal_span": "We introduce WorldTravel, a benchmark comprising 150 real-world travel scenarios across 5 cities that demand navigating an average of 15+ interdependent temporal and logical constraints.",
  "horizon_span": "We identify a critical Perception-Action Gap and a Planning Horizon threshold at approximately 10 constraints where model reasoning consistently fails",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285452575",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "other (singleton)",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "281892403",
  "title": "COMPASS: Benchmarking Constrained Optimization in LLM Agents",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 6,
  "publication_date": "2025-10-08",
  "months_since_pub": 11,
  "citations_per_month": 0.55,
  "artifact_name": "COMPASS",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "gather task information and constraints from the user via multi-turn conversation; use tools to gather relevant information from a database; propose a travel plan satisfying all hard constraints; optimize the plan for the user's utility objective beyond mere feasibility",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Constraint satisfaction and utility optimization are coupled: gathering more information via conversation/tools expands the feasible search space, and success at optimizing utility strongly correlates with how much information was gathered, so under-exploring earlier in the conversation directly limits achievable optimality later.",
  "n_goals": null,
  "tracking_demand": "Agent must track constraints and preferences gathered so far via conversation and tool calls, the current feasible-solution search space, and how well a candidate plan satisfies both hard constraints and the utility objective.",
  "scoring": "other:feasibility-vs-optimality-gap - reports feasibility (constraint satisfaction, 70-90%) and optimality (utility optimization, 20-60%) as two separate metrics revealing a significant feasible-optimal gap; a two-dimensional scoring scheme rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "State-of-the-art models achieve 70-90% feasibility but only 20-60% optimality, revealing a significant feasible-optimal gap; success correlates with information gathered rather than tool-use ability; no explicit human/expert baseline is given.",
  "availability": null,
  "goal_span": "To success in these tasks, agents must engage in multi-turn conversations with user to gather task information as well as use tools to gather information from the database. Then agents must propose a solution that not only satisfies hard constraints but also optimizes user's utility objective.",
  "horizon_span": "To success in these tasks, agents must engage in multi-turn conversations with user to gather task information as well as use tools to gather information from the database.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:281892403",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282748914",
  "title": "CostBench: Evaluating Multi-Turn Cost-Optimal Planning and Adaptation in Dynamic Environments for LLM Tool-Use Agents",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 30,
  "publication_date": "2025-11-04",
  "months_since_pub": 10,
  "citations_per_month": 3.0,
  "artifact_name": "CostBench",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "find a cost-optimal sequence of atomic and composite tool calls to solve a travel-planning task; detect and adapt to dynamic blocking events (e.g. tool failures, cost changes) that occur mid-task; replan in real time to remain cost-optimal after a blocking event",
  "goal_origin": "mixed:given-up-front-task-with-environment-injected-blocking-events",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are 'solvable via multiple sequences of atomic and composite tools with diverse, customizable costs', so an agent's tool-sequence choice constrains total cost, and dynamic blocking events (tool failures, cost changes) injected mid-task can invalidate the current plan, forcing dependent replanning.",
  "n_goals": null,
  "tracking_demand": "Agent must track the accumulated cost of its chosen tool sequence so far, remaining budget, and whether any of four types of dynamic blocking events have occurred, requiring real-time replanning to stay cost-optimal.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports an 'exact match rate' for cost-optimal solutions (below 75% for GPT-5 on hardest static tasks, dropping ~40% further under dynamic conditions), which is closer to binary-final-success (was the cost-optimal plan found) than to explicit partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even GPT-5 achieves less than 75% exact-match rate on the hardest static tasks, with performance dropping by around 40% under dynamic conditions (no human baseline given).",
  "availability": null,
  "goal_span": "It also supports four types of dynamic blocking events, such as tool failures and cost changes, to simulate real-world unpredictability and necessitate agents to adapt in real time.",
  "horizon_span": "CostBench comprises tasks solvable via multiple sequences of atomic and composite tools with diverse, customizable costs",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282748914",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "mixed",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "279243068",
  "title": "Flex-TravelPlanner: A Benchmark for Flexible Planning with Language Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 11,
  "publication_date": "2025-06-05",
  "months_since_pub": 15,
  "citations_per_month": 0.73,
  "artifact_name": "Flex-TravelPlanner",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "revise a travel plan as new constraints are introduced sequentially across turns; correctly prioritize competing constraints when a newly introduced lower-priority preference conflicts with an existing higher-priority constraint",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Later-introduced constraints can conflict with earlier ones and must be prioritized correctly rather than simply appended, so plan revision at each turn depends on correctly weighing all previously introduced constraints together.",
  "n_goals": null,
  "tracking_demand": "The agent must track which constraints have been introduced so far, their relative priority, and whether previously satisfied requirements are still respected as new constraints arrive turn by turn.",
  "scoring": "other:multi-turn-constraint-adherence-analysis. The abstract reports comparative performance across single-turn vs multi-turn settings and analyzes constraint-prioritization errors, but does not describe an explicit subgoal-checkpoint or milestone rubric.",
  "horizon_value": "up to 3 turns (all-at-once / 2-turn / 3-turn constraint-introduction patterns) across 120 base queries",
  "horizon_unit": "turns",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per query/plan -- each of the 120 base queries is evaluated under each turn-count condition",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": "https://github.com/juhyunohh/FlexTravelBench",
  "goal_span": "we introduce two novel evaluation settings: (1) sequential constraint introduction across multiple turns, and (2) scenarios with explicitly prioritized competing constraints ... models struggle with constraint prioritization, often incorrectly favoring newly introduced lower priority preferences over existing higher-priority constraints.",
  "horizon_span": "all-at-once (N), 2-turn (N-1, 1), and 3-turn (N-2, 1, 1) scenarios ... we construct multi-turn scenarios using 120 queries from TravelPlanner's validation set",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279243068",
  "provenance": "asta-find,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "turns",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280700508",
  "title": "RETAIL: Towards Real-world Travel Planning for Large Language Models",
  "year": 2025,
  "venue": "Conference on Empirical Methods in Natural Language Processing",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": "business-office-enterprise",
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 9,
  "publication_date": "2025-08-21",
  "months_since_pub": 13,
  "citations_per_month": 0.69,
  "artifact_name": "RETAIL",
  "artifact_kind": "dataset",
  "domain": "travel-and-constraint-planning",
  "goal_types": "infer and satisfy implicit user requirements (not just explicit queries); satisfy explicit queries with or without later revision needs; account for diverse environmental factors and constraints to ensure plan feasibility; produce an all-in-one plan with rich, detailed POI (point-of-interest) arrangement rather than only basic POI listing",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Environmental awareness constraints (e.g., feasibility given real-world factors) interact with the detailed POI arrangement and any needed revision, so satisfying one requirement (e.g., adding a detailed POI) can violate feasibility or another implicit constraint, requiring the plan to jointly satisfy all of them.",
  "n_goals": null,
  "tracking_demand": "The agent must infer unstated (implicit) requirements, track environmental constraints affecting plan feasibility, and assemble detailed POI information into a single all-in-one plan, revising an existing plan when a revision need is present.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even the strongest existing model achieves only a 1.0% pass rate on RETAIL; the authors' TGMA framework improves this to 2.72%. No human/expert baseline given.",
  "availability": null,
  "goal_span": "Our experiments reveal that even the strongest existing model achieves merely a 1.0% pass rate, indicating real-world travel planning remains extremely challenging. In contrast, TGMA demonstrates substantially improved performance 2.72%",
  "horizon_span": "Second, existing solutions ignore diverse environmental factors and user preferences, limiting the feasibility of plans.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280700508",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279403144",
  "title": "Wide-Horizon Thinking and Simulation-Based Evaluation for Real-World LLM Planning with Multifaceted Constraints",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "in",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 5,
  "publication_date": "2025-06-14",
  "months_since_pub": 15,
  "citations_per_month": 0.33,
  "artifact_name": "Travel-Sim",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "satisfy multiple parallel, potentially conflicting real-world planning constraints (e.g. preferences, logistics) within one itinerary; respect causal dependencies where earlier itinerary choices constrain which later activities remain feasible; produce a plan validated via realistic agent-based simulation rather than isolated constraint checks",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Earlier itinerary choices causally constrain which later activities remain feasible, so constraints cannot be satisfied independently; the plan must respect this ordering rather than checking each constraint in isolation.",
  "n_goals": null,
  "tracking_demand": "The planner must track all outstanding multifaceted constraints and the downstream causal consequences of already-committed itinerary decisions as the simulated trip unfolds.",
  "scoring": "other:simulation-based-plan-assessment. The abstract states plans are assessed via agent-based simulation that 'inherently resolves' causal dependencies among constraints, rather than describing an explicit subgoal-checkpoint or milestone rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we propose Travel-Sim, an agent-based benchmark assessing plans via real-world simulation, thereby inherently resolving these causal dependencies.",
  "horizon_span": "we propose Travel-Sim, an agent-based benchmark assessing plans via real-world simulation, thereby inherently resolving these causal dependencies.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279403144",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284313793",
  "title": "Beyond Itinerary Planning-A Real-World Benchmark for Multi-Turn and Tool-Using Travel Tasks",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 6,
  "publication_date": "2025-12-27",
  "months_since_pub": 9,
  "citations_per_month": 0.67,
  "artifact_name": "TravelBench",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "solve a travel-planning problem independently using cached tool results; interact with the user across multiple turns to elicit implicit preferences; correctly recognize and communicate the agent's own capability boundaries (Unsolvable subtask)",
  "goal_origin": "mixed:given-up-front-plus-injected-by-user-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "The three subtasks (Single-Turn, Multi-Turn, Unsolvable) probe distinct capabilities, but within the Multi-Turn subtask a later turn's preference elicitation depends on correctly tracking what was already stated in earlier turns, and all three rely on a shared sandbox of ten cached travel tools.",
  "n_goals": "three subtasks (Single-Turn, Multi-Turn, Unsolvable)",
  "tracking_demand": "In the Multi-Turn subtask, the agent must track previously elicited or stated user preferences across turns and integrate them with cached tool results from a sandbox of ten travel-related tools, while also recognizing when a request exceeds its capability boundaries.",
  "scoring": "other:per-subtask-capability-imbalance-analysis",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "we construct three subtasks -- $\\textit{Single-Turn}$, $\\textit{Multi-Turn}$, and $\\textit{Unsolvable}$ -- to evaluate agents' three core capabilities in real settings: (1) solving problems independently, (2) interacting with users to elicit implicit preferences, and (3) recognizing the capability boundaries. ... we evaluate multiple LLMs on TravelBench and find that even advanced models exhibit imbalanced performance across different capabilities.",
  "horizon_span": "we cache real tool-call results and build a sandbox environment which integrates ten travel-related tools",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284313793",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276724727",
  "title": "TripCraft: A Benchmark for Spatio-Temporally Fine Grained Travel Planning",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 31,
  "publication_date": "2025-02-27",
  "months_since_pub": 19,
  "citations_per_month": 1.63,
  "artifact_name": "TripCraft",
  "artifact_kind": "dataset",
  "domain": "travel-and-constraint-planning",
  "goal_types": "generate a spatiotemporally coherent 7-day travel itinerary; satisfy meal-scheduling constraints (Temporal Meal Score); satisfy attraction-timing constraints (Temporal Attraction Score); satisfy spatial feasibility across the itinerary (Spatial Score); satisfy activity-ordering constraints (Ordering Score); satisfy user-persona preferences (Persona Score)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "The five evaluation dimensions (meal timing, attraction timing, spatial feasibility, ordering, persona) are jointly evaluated on a single generated 7-day itinerary, so satisfying one (e.g., attraction timing) can conflict with another (e.g., spatial feasibility) since all activities in the plan share the same limited time and space.",
  "n_goals": "five continuous evaluation metrics (Temporal Meal Score, Temporal Attraction Score, Spatial Score, Ordering Score, Persona Score) over a 7-day itinerary",
  "tracking_demand": "The agent must track public-transit schedules, event availability, attraction categories, and user-persona preferences simultaneously while assembling a spatially and temporally consistent 7-day itinerary.",
  "scoring": "other:five-continuous-evaluation-metrics(non-binary). The paper explicitly moves beyond existing binary validation methods by proposing five continuous evaluation metrics assessing itinerary quality across multiple dimensions -- explicit non-binary, multi-axis partial credit rather than one pass/fail label.",
  "horizon_value": "7",
  "horizon_unit": "simulated-days",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode/task (one full generated itinerary spans 7 days)",
  "horizon_stated": "yes",
  "headline_result": "A parameter-informed setting improves the Temporal Meal Score from 61% to 80% in a 7-day scenario; no single top model or human baseline is named in the abstract.",
  "availability": null,
  "goal_span": "To evaluate LLM generated plans beyond existing binary validation methods, we propose five continuous evaluation metrics, namely Temporal Meal Score, Temporal Attraction Score, Spatial Score, Ordering Score, and Persona Score which assess itinerary quality across multiple dimensions.",
  "horizon_span": "improving the Temporal Meal Score from 61% to 80% in a 7 day scenario",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276724727",
  "provenance": "asta-find,parametric",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "simulated-days",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "282056297",
  "title": "TripScore: Benchmarking and rewarding real-world travel planning with fine-grained evaluation",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 9,
  "publication_date": "2025-10-10",
  "months_since_pub": 11,
  "citations_per_month": 0.82,
  "artifact_name": "TripScore",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "produce a travel itinerary that jointly satisfies fine-grained feasibility, reliability, and engagement criteria; achieve a high unified reward score combining these criteria for RL training/evaluation; generalize to real-world, free-form travel requests",
  "goal_origin": "given-up-front",
  "decomposition": "other:joint-constraint-satisfaction-within-one-itinerary",
  "interdependence": "Feasibility, reliability, and engagement criteria are unified into a single reward, so improving one dimension (e.g., engagement) must not come at the cost of another (e.g., feasibility) for the overall reward to increase.",
  "n_goals": null,
  "tracking_demand": "Agent must track and jointly satisfy fine-grained feasibility, reliability, and engagement criteria while assembling a travel plan, since these are unified into a single reward score used for both evaluation and RL training.",
  "scoring": "continuous-reward",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The evaluator achieves 60.75% agreement with travel-expert annotations, outperforming multiple LLM-as-judge baselines; RL via GRPO generally improves itinerary feasibility over prompt-only and supervised baselines, yielding higher unified reward scores; no single top-line agent-vs-human success rate is given.",
  "availability": null,
  "goal_span": "We introduce a comprehensive benchmark for travel planning that unifies fine-grained criteria into a single reward, enabling direct comparison of plan quality and seamless integration with reinforcement learning (RL).",
  "horizon_span": "We further release a large-scale dataset of 4,870 queries including 219 real-world, free-form requests for generalization to authentic user intent.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282056297",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "continuous / cumulative reward",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282384587",
  "title": "TripTide: A Benchmark for Adaptive Travel Planning under Disruptions",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 5,
  "publication_date": "2025-10-24",
  "months_since_pub": 11,
  "citations_per_month": 0.45,
  "artifact_name": "TripTide",
  "artifact_kind": "benchmark",
  "domain": "travel-and-constraint-planning",
  "goal_types": "preserve original itinerary intent (feasibility and goals) after a disruption; respond promptly and appropriately to a disruption event (flight cancellation, weather closure, overbooked attraction); adapt the itinerary with appropriate semantic, spatial, and sequential divergence from the original plan; maintain plan quality across varying disruption severity and traveler tolerance levels",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "A disruption forces revision of a previously constructed itinerary while preserving as much of its original intent, spatial coherence, and sequential/timing structure as possible, so the revised plan's later legs remain constrained by decisions made earlier in the original (pre-disruption) plan.",
  "n_goals": null,
  "tracking_demand": "The agent must track the original itinerary's intent, spatial layout, and sequential structure, detect and appropriately size its response to a disruption of a given severity and traveler tolerance, and measure how much the revision diverges from the original across semantic, spatial, and sequential dimensions.",
  "scoring": "other:automatic-metrics-plus-LLM-judge-plus-manual-expert-evaluation",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "LLMs maintain strong sequential consistency and semantic stability under disruption, but disruption-handling ability declines as plan length increases; no single specific top-model score or human-expert baseline percentage is given in the abstract.",
  "availability": null,
  "goal_span": "we introduce automatic metrics including Preservation of Intent (how well the revised plan maintains feasibility and goals), Responsiveness (promptness and appropriateness of disruption handling), and Adaptability (semantic, spatial, and sequential divergence between original and revised plans).",
  "horizon_span": "disruption-handling ability declines as plan length increases, highlighting limits in LLM robustness",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282384587",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "LLM-judge or expert rubric",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "278789269",
  "title": "Is Your LLM-Based Multi-Agent a Reliable Real-World Planner? Exploring Fraud Detection in Travel Planning",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "travel-and-constraint-planning",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "travel-and-constraint-planning",
  "citation_count": 19,
  "publication_date": "2025-05-22",
  "months_since_pub": 16,
  "citations_per_month": 1.19,
  "artifact_name": "WandaPlan",
  "artifact_kind": "environment/simulator",
  "domain": "travel-and-constraint-planning",
  "goal_types": "produce a correct multi-agent travel plan while resisting injected deceptive/fraudulent content; detect/avoid Misinformation Fraud; detect/avoid Team-Coordinated Multi-Person Fraud; detect/avoid Level-Escalating Multi-Round Fraud",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "set-of-independent",
  "interdependence": "In Level-Escalating Multi-Round Fraud and Team-Coordinated Multi-Person Fraud specifically, deceptive content compounds or coordinates across multiple rounds/agents, so failing to catch an earlier piece of fraud can let it escalate or be reinforced by later coordinated fraudulent inputs.",
  "n_goals": "three fraud cases (Misinformation, Team-Coordinated Multi-Person, Level-Escalating Multi-Round)",
  "tracking_demand": "The planning system must track which review/social-media sourced information it has incorporated, cross-check it for authenticity across rounds and multiple purported sources, and avoid building a travel plan on fraudulent inputs even as fraud escalates over multiple rounds.",
  "scoring": "other:framework-weakness-analysis-across-three-fraud-cases",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Existing multi-agent planning frameworks show significant weaknesses across all three fraud cases, prioritizing task efficiency over data authenticity; the authors' proposed anti-fraud agent is offered as a mitigation, but no single headline accuracy number or human baseline is given in the abstract.",
  "availability": null,
  "goal_span": "We assess system performance across three fraud cases: Misinformation Fraud, Team-Coordinated Multi-Person Fraud, and Level-Escalating Multi-Round Fraud. We reveal significant weaknesses in existing frameworks that prioritize task efficiency over data authenticity.",
  "horizon_span": "Level-Escalating Multi-Round Fraud",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278789269",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "285606721",
  "title": "AgenticShop: Benchmarking Agentic Product Curation for Personalized Web Shopping",
  "year": 2026,
  "venue": "The Web Conference",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain-other",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "other:personalized-web-shopping",
  "citation_count": 10,
  "publication_date": "2026-02-12",
  "months_since_pub": 7,
  "citations_per_month": 1.43,
  "artifact_name": "AgenticShop",
  "artifact_kind": "benchmark",
  "domain": "other:personalized-web-shopping",
  "goal_types": "explore the open web to curate a set of products satisfying diverse shopping scenarios; satisfy each checklist item in a verifiable, checklist-driven personalization rubric aligned to a user profile",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Not detailed beyond joint personalization fit across checklist items; no explicit ordering, resource-sharing, or exclusivity constraints are described in the abstract.",
  "n_goals": null,
  "tracking_demand": "The agent must track a user's personalization checklist criteria and diverse profile preferences while exploring open-web shopping scenarios, so curated products satisfy every checklist item.",
  "scoring": "milestone-rubric -- a checklist-driven personalization evaluation framework verifies satisfaction of individual criteria, a form of subgoal-level (checklist-item) credit rather than a single binary pass/fail.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Crucially, our approach features realistic shopping scenarios, diverse user profiles, and a verifiable, checklist-driven personalization evaluation framework.",
  "horizon_span": "Crucially, our approach features realistic shopping scenarios, diverse user profiles, and a verifiable, checklist-driven personalization evaluation framework.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285606721",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "286670082",
  "title": "AndroTMem: From Interaction Trajectories to Anchored Memory in Long-Horizon GUI Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 11,
  "publication_date": "2026-03-19",
  "months_since_pub": 6,
  "citations_per_month": 1.83,
  "artifact_name": "AndroTMem-Bench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "carry forward critical intermediate state across a long sequence of GUI interaction steps to complete a task; correctly resolve strong step-to-step causal dependencies where sparse intermediate states are decisive for later actions; complete each of 1,069 Android GUI tasks (avg. 32.1 steps, max. 65 steps)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are specifically designed so that sparse but essential intermediate states are decisive for downstream actions, meaning correctness of later steps causally depends on correctly carrying forward specific earlier intermediate states rather than merely on recent context.",
  "n_goals": "1,069 tasks with 34,473 total interaction steps (avg. 32.1 per task, max. 65)",
  "tracking_demand": "The agent must carry forward sparse, dependency-critical intermediate state across an average of 32.1 (up to 65) interaction steps per task, since full-sequence replay is redundant/noisy and naive summarization erases exactly the dependency-critical information needed for later steps.",
  "scoring": "other:task-complete-rate-TCR-plus-AMS-memory-attribution-metric",
  "horizon_value": "1,069 tasks with 34,473 total interaction steps; average 32.1 steps per task, maximum 65 steps per task",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (a single Android GUI interaction task)",
  "horizon_stated": "yes",
  "headline_result": "Across 12 evaluated GUI agents, the proposed Anchored State Memory (ASM) improves Task Complete Rate by 5%-30.16% and AMS by 4.93%-24.66% over full-sequence replay and summary-based baselines; no explicit human/expert baseline is given (comparison is across memory methods).",
  "availability": "https://github.com/CVC2233/AndroTMem",
  "goal_span": "Its core benchmark, AndroTMem-Bench, comprises 1,069 tasks with 34,473 interaction steps (avg. 32.1 per task, max. 65). We evaluate agents with TCR (Task Complete Rate), focusing on tasks whose completion requires carrying forward critical intermediate state.",
  "horizon_span": "Its core benchmark, AndroTMem-Bench, comprises 1,069 tasks with 34,473 interaction steps (avg. 32.1 per task, max. 65).",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:286670082",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288701440",
  "title": "AndroidDaily: A Verifiable Benchmark for Mobile GUI Agents on Real-World Closed-Source Applications",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 4,
  "publication_date": "2026-05-26",
  "months_since_pub": 4,
  "citations_per_month": 1.0,
  "artifact_name": "AndroidDaily",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete each of 350 realistic daily-use tasks spanning 94 real closed-source Android apps; satisfy step-level operational obligations per GRADE's guideline criteria; meet output-quality criteria at each step; avoid violating negative constraints at each step",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "GRADE tracks the agent's visual trajectory step-by-step against three tiers of guidelines (obligations, quality, negative constraints); correctly satisfying later steps depends on the state produced by earlier steps in the closed-source, non-instrumented app.",
  "n_goals": "350 tasks across 94 apps; three guideline tiers per task (operational obligations, output quality, negative constraints)",
  "tracking_demand": "The agent must track progress against multiple step-level guideline criteria (obligations, quality, negative constraints) across a long-horizon, open-ended interaction with a closed-source app exposing no internal state.",
  "scoring": "subgoal-checkpoint-partial-credit. GRADE explicitly produces step-level diagnostic judgments against observable guidelines (87.37% agreement with human evaluators), i.e., process-level/step credit exists alongside the overall 62.0% best-model success rate.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "The strongest model reaches a 62.0% success rate on AndroidDaily; GRADE itself achieves 87.37% agreement with human evaluators; no human baseline for task success itself is given.",
  "availability": null,
  "goal_span": "GRADE tracks the agent's visual trajectory against these criteria and produces step-level diagnostic judgments, turning long-horizon, open-ended mobile interactions into verifiable evaluation without relying on hidden internal states.",
  "horizon_span": "turning long-horizon, open-ended mobile interactions into verifiable evaluation without relying on hidden internal states",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288701440",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "290990615",
  "title": "CAP: A Scalable Benchmark for Evaluating Cross-Site Browser Agents with Complex Actions and Perception",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-08-09",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "CAP",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete a realistic cross-site workflow requiring several specific operations on each of multiple real-world websites; correctly interact with complex, dynamically rendered UI elements (non-trivial UI interactions); correctly perceive/interpret dynamically rendered visual content across the workflow",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Each task is grounded in multiple specific operations across several websites recomposed via a decomposition-and-recomposition pipeline, so completing a later cross-site step depends on having correctly completed the operation(s) on the preceding site(s) in the workflow.",
  "n_goals": "420 tasks across 108 real-world websites and 24 domains; average of 7 execution points and 4 perception points per task",
  "tracking_demand": "The agent must track which execution and perception checkpoints it has already satisfied across multiple websites within one recomposed cross-site workflow, using a verifiable agent-as-a-judge evaluation framework.",
  "scoring": "milestone-rubric. The benchmark's 'verifiable agent-as-a-judge evaluation framework' scores against explicit execution and perception checkpoints (avg. 7 execution + 4 perception points per task), i.e. genuine per-checkpoint (subgoal-level) partial credit rather than only end-to-end success.",
  "horizon_value": "baselines evaluated with a maximum of 50 reasoning-action steps per task; average of 7 execution points and 4 perception points per task",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we construct 420 tasks across 108 real-world websites and 24 domains under careful quality control. Experiments on state-of-the-art browser agents using our verifiable agent-as-a-judge evaluation framework show low success rates and reveal that perception-heavy interactions remain a major bottleneck.",
  "horizon_span": "baselines were evaluated with a maximum of 50 reasoning-action steps per task. Additionally, the benchmark has an average of 7 execution points and 4 perception points per task.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290990615",
  "provenance": "forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287256395",
  "title": "ClawBench: Can AI Agents Complete Everyday Online Tasks?",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 22,
  "publication_date": "2026-04-09",
  "months_since_pub": 5,
  "citations_per_month": 4.4,
  "artifact_name": "ClawBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete everyday online tasks (purchases, appointment bookings, job applications) across 144 real platforms; obtain relevant information from user-provided documents; navigate multi-step workflows across diverse platforms; correctly fill in many detailed form fields per task",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Correctly filling later form fields often depends on information obtained earlier from user-provided documents, and each platform's workflow steps must be completed in the required order.",
  "n_goals": "153 tasks across 144 platforms and 15 categories",
  "tracking_demand": "Agent must extract and correctly carry information from user-provided documents through a multi-step, write-heavy workflow with many detailed form fields, operating on live, dynamic production websites.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Claude Sonnet 4.6 achieves only 33.3% task success, the best among 8 evaluated frontier models; no human/expert baseline reported.",
  "availability": null,
  "goal_span": "ClawBench, an evaluation framework comprising 153 everyday online tasks that people need to accomplish regularly in their lives and work, spanning 144 platforms across 15 categories, from completing purchases and booking appointments to submitting job applications. These tasks require capabilities beyond existing benchmarks, such as obtaining relevant information from user-provided documents, navigating multi-step workflows across diverse platforms, and write-heavy operations like filling in many detailed forms correctly.",
  "horizon_span": "153 everyday online tasks that people need to accomplish regularly in their lives and work, spanning 144 platforms across 15 categories",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287256395",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290991102",
  "title": "ComboShoppingBench: Evaluating LLM Agents for Budget-Constrained Basket Shopping with Coupons",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain-other",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "other:commerce-combo-shopping",
  "citation_count": 0,
  "publication_date": "2026-08-10",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "ComboShoppingBench",
  "artifact_kind": "benchmark",
  "domain": "other:commerce-combo-shopping",
  "goal_types": "construct a basket of complementary items satisfying compatibility constraints; keep the basket within a stated budget while optimizing coupon use; satisfy store-level requirements, availability, and delivery fees jointly",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Item choices are mutually constrained by compatibility, shared budget, coupon eligibility (which may depend on which other items are in the basket), and store-level rules, so adding or swapping one item can invalidate the whole basket.",
  "n_goals": null,
  "tracking_demand": "Agent must track the running basket contents, remaining budget, which coupons are valid given the current basket, and per-store requirements while searching for a feasible combination.",
  "scoring": "other:mixed \u2014 LLM judges assess semantic satisfaction/response quality/claim faithfulness while deterministic validators separately check product-ID validity, budget compliance, and coupon optimality; this is closer to milestone-rubric/LLM-judge-rubric combined than a single binary outcome.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Experiments show 'even strong agents struggle on ComboShoppingBench' (no specific numeric score or human baseline given in the abstract).",
  "availability": null,
  "goal_span": "requiring joint reasoning about item compatibility, availability, store-level requirements, delivery fees, coupons, and budgets",
  "horizon_span": "We introduce ComboShoppingBench, an agentic shopping benchmark for open-ended yet verifiable basket construction in a simulated commerce and takeout environment.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290991102",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289669596",
  "title": "DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-06-25",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "DMV-Bench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete chains of autonomous shopping sessions (browsing/selecting home-furnishing products); recall a unique, pre-rendered incidental visual cue seen earlier on a product image when later asked about it",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "Later recall queries depend on visual cues incidentally observed (not deliberately recorded) during earlier shopping sessions in the chain, so success requires carrying forward visual information the agent never explicitly chose to remember.",
  "n_goals": null,
  "tracking_demand": "Agent must retain visual memory of incidental cues (not just deliberately extracted facts) across chains of shopping sessions of varying length, without being told in advance which details will later be tested.",
  "scoring": "other:not-stated \u2014 the abstract reports comparative performance of memory architectures (DualMem vs. caption-only and other agent-memory baselines) but does not describe a partial-credit/checkpoint scheme for individual recall queries.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "DualMem outperforms a caption-only baseline and three recent multimodal agent-memory systems across multi-session chain lengths on multiple models (no single numeric headline score or human baseline given).",
  "availability": "https://github.com/yyyujintang/DMV-Bench",
  "goal_span": "agents undergo chains of autonomous shopping sessions in which every visited product image carries a unique, pre-rendered incidental cue that the agent is later asked to recall",
  "horizon_span": "DualMem outperforms a caption-only baseline and three recent multimodal agent-memory systems across multi-session chain lengths on multiple models",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289669596",
  "provenance": "forward-citation",
  "scoring_family": "not recorded",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "291470161",
  "title": "Benchmarking General Mobile Assistants in Challenging Real-World Scenarios",
  "year": 2026,
  "venue": "",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-08-21",
  "months_since_pub": 1,
  "citations_per_month": 0.0,
  "artifact_name": "GMA",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete tasks ranging from atomic actions to complex multi-step workflows across seven open-source-based applications; handle lifestyle-sharing and travel-planning domains at four escalating difficulty tiers; maintain context/state across a workflow via harness-level context retention and explicit state tracking",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Higher-tier, multi-step workflows require maintaining context and explicit state across steps, and controlled ablations show harness-level context retention and state tracking materially affect the ability to complete demanding, dependent workflows.",
  "n_goals": "300 tasks across four difficulty tiers and seven applications",
  "tracking_demand": "Agent must maintain context retention and explicit state tracking across multi-step workflows spanning seven applications, since performance declines substantially as task complexity/tier increases.",
  "scoring": "other:success-rate-by-difficulty-tier",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "GMA introduces seven applications based on open-source projects, spanning domains such as lifestyle sharing and travel planning, and 300 tasks across four difficulty tiers, from atomic actions to complex multi-step workflows.",
  "horizon_span": "300 tasks across four difficulty tiers, from atomic actions to complex multi-step workflows",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:291470161",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "288741589",
  "title": "GTA: Generating Long-horizon Tasks for Web Agents at Scale",
  "year": 2026,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 0.0,
  "artifact_name": "GTA",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete multi-hop, cross-page web tasks that are compositional over a site graph; follow an intermediate trajectory of dense, process-level supervision steps, not just reach a coarse end goal; generalize across more than 50 websites (e-commerce, government, forums, news) including multilingual tasks",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are explicitly grounded in a site graph so that multi-hop steps compose across pages; each hop's completion (validated via deterministic replay) is a precondition for correctly reaching the next page/step in the trajectory.",
  "n_goals": null,
  "tracking_demand": "The agent must track its position and accumulated information across a multi-hop, cross-page trajectory grounded in a site graph, since tasks provide dense process-level supervision (intermediate trajectory steps) rather than only a coarse start-goal annotation.",
  "scoring": "other:process-level-diagnostics-plus-human-agent-performance-gap",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The resulting benchmark reveals a significant human-agent performance gap, though the abstract does not state an exact percentage figure or specific model name for the top performer.",
  "availability": null,
  "goal_span": "This design decouples crawling from generation for greater efficiency, grounds tasks in the site graph to enforce compositionality, and ensures dense supervision through deterministic replays and systematic validation... The resulting benchmark reveals a significant human-agent performance gap and enables detailed diagnostics.",
  "horizon_span": "These limitations prevent reliable training and evaluation of agents that must generalize to realistic, multi-hop, cross-page tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288741589",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287023445",
  "title": "When Users Change Their Mind: Evaluating Interruptible Agents in Long-Horizon Web Navigation",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 6,
  "publication_date": "2026-04-01",
  "months_since_pub": 5,
  "citations_per_month": 1.2,
  "artifact_name": "InterruptBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "execute a long-horizon, environmentally grounded web-navigation task; adapt when the user adds a new requirement mid-task; revise the current goal when the user changes an existing requirement mid-task; abandon a sub-goal when the user retracts a requirement mid-task; recover efficiently without redundant or incorrect actions after an interruption",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Actions induce persistent state changes in the web environment, and mid-task interruptions (addition, revision, retraction) must be reconciled with already-taken actions, so later steps are constrained by both the (possibly revised) goal and irreversible prior state changes.",
  "n_goals": null,
  "tracking_demand": "The agent must track the current, possibly revised, task intent across single- and multi-turn interruption settings plus the environment's persistent state from already-executed actions to adapt or recover correctly.",
  "scoring": "other:effectiveness-and-efficiency-of-adaptation-and-recovery. No itemized subgoal-checkpoint rubric is described; the paper instead scores agents along two axes (adaptation effectiveness, recovery efficiency) rather than a single binary or milestone score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": null,
  "availability": "https://github.com/HenryPengZou/InterruptBench",
  "goal_span": "We formalize three realistic interruption types, including addition, revision, and retraction, and introduce InterruptBench, a benchmark derived from WebArena-Lite that synthesizes high-quality interruption scenarios under strict semantic constraints.",
  "horizon_span": "the first systematic study of interruptible agents in long-horizon, environmentally grounded web navigation tasks, where actions induce persistent state changes.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287023445",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288256230",
  "title": "LongMemEval-V2: Evaluating Long-Term Agent Memory Toward Experienced Colleagues",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 9,
  "publication_date": "2026-05-12",
  "months_since_pub": 4,
  "citations_per_month": 2.25,
  "artifact_name": "LongMemEval-V2 (LME-V2)",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "recall static state facts about the environment (static state recall); track dynamic state changes over time (dynamic state tracking); recall workflow knowledge/procedures learned from experience; recall environment-specific 'gotchas'/recurring failure modes; maintain premise awareness of what has already been established or assumed",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "set-of-independent",
  "interdependence": "All five memory abilities are tested against the same underlying history trajectories (up to 500 trajectories, 115M tokens), so correctly answering later questions depends on having internalized relevant environment-specific experience accumulated across many earlier trajectories.",
  "n_goals": "451 manually curated questions covering five core memory abilities; history trajectories of up to 500 trajectories and 115M tokens",
  "tracking_demand": "The memory system must consume and internalize up to 500 history trajectories (115M tokens) and return compact, correct evidence for downstream question answering across five distinct memory-ability categories.",
  "scoring": "other:per-question-accuracy-across-five-memory-abilities. Accuracy is reported per memory-ability category (e.g., AgentRunbook-C reaches 72.5% average accuracy vs. 48.5% for the strongest RAG baseline and 69.3% for the off-the-shelf coding-agent baseline), a per-competency (subgoal-type) breakdown beyond one aggregate score.",
  "horizon_value": "up to 500 trajectories and 115M tokens of history",
  "horizon_unit": "episodes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per whole episode (the full history-trajectory context a memory system must consume before answering a question)",
  "horizon_stated": "yes",
  "headline_result": "AgentRunbook-C achieves 72.5% average accuracy, outperforming the strongest RAG baseline (48.5%) and the off-the-shelf coding-agent baseline (69.3%), though at higher latency cost; no human/expert baseline given.",
  "availability": null,
  "goal_span": "LME-V2 contains 451 manually curated questions covering five core memory abilities for web agents: static state recall, dynamic state tracking, workflow knowledge, environment gotchas, and premise awareness. Questions are paired with history trajectories containing up to 500 trajectories and 115M tokens.",
  "horizon_span": "Questions are paired with history trajectories containing up to 500 trajectories and 115M tokens.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288256230",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "episodes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288741395",
  "title": "STAMP: Training Explicit Memory for Mobile GUI Agents in Controllable and Scalable Virtual Environments",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 1,
  "publication_date": "2026-05-28",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "Memory-World",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "encode a programmatically-injected memory variable at the correct point in a task; retain that memory variable correctly despite progressive discarding of older visual history; retrieve and correctly apply the memorized variable later in the same long-horizon mobile GUI task",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Deterministic memory variables are programmatically injected to control what must be memorized, when it should be encoded, and when it must later be retrieved, so correct retrieval at a later point in the task strictly depends on correct encoding of the injected variable at an earlier point, especially as older visual history is discarded to save context.",
  "n_goals": null,
  "tracking_demand": "The agent must explicitly memorize deterministic variables injected at specific points and correctly retrieve/apply them later in the same task, despite token-heavy screenshots forcing progressive discarding of older visual history.",
  "scoring": "other:memory-accuracy-and-task-resilience-vs-gui-specialized-baselines",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Stamp-GUI achieves state-of-the-art performance among GUI-specialized models and sets a new high watermark on Memory-World, while maintaining strong general mobile navigation capabilities; no human baseline given.",
  "availability": null,
  "goal_span": "STAMP, a framework that trains explicit memory in mobile agents through controllable virtual environments, where deterministic memory variables are programmatically injected into synthesized tasks to control what must be memorized, when it should be encoded, and when it must later be retrieved ... Evaluated on our newly introduced Memory-World benchmark, the resulting Stamp-GUI agent achieves state-of-the-art performance",
  "horizon_span": "Mobile GUI agents excel at immediate reactive control but frequently fail in realistic, long-horizon tasks that require memory.",
  "extraction_confidence": 1,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288741395",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287425479",
  "title": "MobiFlow: Real-World Mobile Agent Benchmarking through Trajectory Fusion",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-02-28",
  "months_since_pub": 7,
  "citations_per_month": 0.0,
  "artifact_name": "MobiFlow",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete real-world tasks within arbitrary third-party mobile apps whose success cannot be checked via system-level APIs; have task completion verified via a graph constructed from fusing multiple real user trajectories",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Success verification is based on whether the agent's action sequence matches a valid path through the fused multi-trajectory state graph, so each action's validity depends on the graph node/state reached by prior actions.",
  "n_goals": "240 tasks across 20 third-party applications",
  "tracking_demand": "The agent must track its current position within the app's fused trajectory-state graph and correctly follow one of the valid paths to task completion, since no system-level completion signal is available.",
  "scoring": "other:graph-path-verification -- task completion is verified via alignment with paths in the fused multi-trajectory state graph rather than binary system-state checks; the abstract does not describe finer intra-task subgoal partial credit beyond this graph-based completion check.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "Using an efficient graph-construction algorithm based on multi-trajectory fusion, MobiFlow can effectively compress the state space, support dynamic interaction, and better align with real-world third-party application scenarios. MobiFlow covers 20 widely used third-party applications and comprises 240 diverse real-world tasks, with enriched evaluation metrics.",
  "horizon_span": "MobiFlow covers 20 widely used third-party applications and comprises 240 diverse real-world tasks, with enriched evaluation metrics.",
  "extraction_confidence": 2,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287425479",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "287831597",
  "title": "Odysseys: Benchmarking Web Agents on Realistic Long Horizon Tasks",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 13,
  "publication_date": "2026-04-27",
  "months_since_pub": 5,
  "citations_per_month": 2.6,
  "artifact_name": "Odysseys",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete long-horizon, multi-site web workflows (e.g., comparing products across domains); plan trips across multiple web services; summarize information gathered from multiple search queries; satisfy an average of 6.1 graded rubric criteria per task",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Sub-tasks performed on one site (e.g., gathering a price or itinerary leg) constrain or feed into actions taken on subsequent sites, requiring sustained cross-site context over potentially hours of browsing.",
  "n_goals": "average of 6.1 graded rubrics per task; 200 tasks total",
  "tracking_demand": "The agent must sustain context and accumulated findings across multiple websites and search queries over potentially hours of browsing, and satisfy each of ~6.1 rubric criteria graded per task rather than a single pass/fail check.",
  "scoring": "milestone-rubric",
  "horizon_value": "potentially hours (of browsing); Trajectory Efficiency measured as rubric score per step",
  "horizon_unit": "wall-clock-hours",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (a single long-horizon multi-site web task)",
  "horizon_stated": "yes",
  "headline_result": "The strongest frontier models achieve only a 44.5% success rate and a Trajectory Efficiency of just 1.15% (rubric score per step); no explicit human/expert baseline number is given.",
  "availability": "https://odysseys-website.pages.dev",
  "goal_span": "we introduce Odysseys: a benchmark of 200 long-horizon web tasks derived from real world browsing sessions evaluated on the live Internet. We find that binary pass/fail evaluation is inadequate for long-horizon settings and introduce a rubric-based evaluation, annotating each Odysseys task with an average of 6.1 graded rubrics.",
  "horizon_span": "require sustained context and cross-site reasoning over potentially hours of browsing",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287831597",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "milestone-rubric",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "wall-clock-hours",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "285725843",
  "title": "Learning Personalized Agents from Human Feedback",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain-other",
  "secondary_family": "embodied-household",
  "info_seeking_component": false,
  "domain_detail": "other:embodied-manipulation-and-online-shopping-personalization",
  "citation_count": 15,
  "publication_date": "2026-02-18",
  "months_since_pub": 7,
  "citations_per_month": 2.14,
  "artifact_name": "PAHF benchmarks (embodied manipulation + online shopping)",
  "artifact_kind": "benchmark",
  "domain": "other:embodied-manipulation-and-online-shopping-personalization",
  "goal_types": "learn a new user's initial preferences from scratch via pre-action clarification; ground actions in preferences retrieved from an explicit per-user memory; adapt rapidly to persona shifts using post-action feedback to update memory",
  "goal_origin": "implied-by-constraints",
  "decomposition": "sequential-chain",
  "interdependence": "The three-step loop is inherently sequential: actions grounded in step 2 depend on preferences clarified in step 1, and future iterations of the loop depend on memory updated from step 3's feedback, especially once preferences drift.",
  "n_goals": null,
  "tracking_demand": "Agent must maintain explicit per-user memory across a four-phase protocol, updating stored preferences via dual feedback channels (pre-action clarification, post-action feedback) as it learns initial preferences from scratch and later adapts to persona shifts.",
  "scoring": "other:personalization-error-reduction",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "PAHF operationalizes a three-step loop: (1) seeking pre-action clarification to resolve ambiguity, (2) grounding actions in preferences retrieved from memory, and (3) integrating post-action feedback to update memory when preferences drift. To evaluate this capability, we develop a four-phase protocol and two benchmarks in embodied manipulation and online shopping.",
  "horizon_span": "we develop a four-phase protocol and two benchmarks in embodied manipulation and online shopping. These benchmarks quantify an agent's ability to learn initial preferences from scratch and subsequently adapt to persona shifts.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:285725843",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "290623622",
  "title": "Beyond Sequential Interaction: Benchmarking Parallel Execution and Coordination for GUI Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-07-17",
  "months_since_pub": 2,
  "citations_per_month": 0.0,
  "artifact_name": "ParaGUIBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "identify which GUI sub-tasks can run concurrently despite unstated dependencies; avoid conflicts between concurrent workers modifying shared artifacts; ensure each worker's locally-completed sub-task composes into a globally correct combined result; complete each of 233 tasks spanning six task categories on separate desktop instances",
  "goal_origin": "implied-by-constraints",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Sub-tasks dispatched to concurrent workers share a file system and can conflict when modifying the same artifacts; workers only see their own sub-task, so local completion does not guarantee the combined result satisfies the original instruction -- an explicit mutual-exclusion/shared-resource dependency.",
  "n_goals": "233 tasks across six task categories",
  "tracking_demand": "The planner-worker system must track which sub-tasks are dependency-free and safe to parallelize, coordinate concurrent workers' access to shared artifacts, and verify that the union of workers' outputs satisfies the original combined instruction.",
  "scoring": "other:success-rate-plus-step-reduction-ratio-and-token-cost",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "ParaGUI reaches a 46.4% success rate, outperforming the strongest serial baseline (Claude Sonnet 4.6) by 12.9 points while using roughly half the steps and less than half the tokens; no human baseline given.",
  "availability": null,
  "goal_span": "we introduce ParaGUIBench, to our knowledge, the first benchmark dedicated to parallel execution and coordination of multiple GUI agents on separate desktop instances. It consists of three components: a multi-device Docker infrastructure with a shared file system; a dataset of 233 tasks spanning six task categories; and an evaluation system with efficiency metrics, including step reduction ratio and token cost.",
  "horizon_span": "ParaGUI reaches a 46.4% success rate, outperforming the strongest serial baseline (Claude Sonnet 4.6) by 12.9 points while using roughly half the steps and less than half the tokens.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:290623622",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "implied-by-constraints",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289621498",
  "title": "PhoneBuddy: Training Open Models for Agentic Phone Use",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2026-06-22",
  "months_since_pub": 3,
  "citations_per_month": 0.0,
  "artifact_name": "PhoneWorld",
  "artifact_kind": "environment/simulator",
  "domain": "web/GUI",
  "goal_types": "complete single-app phone tasks; complete mini-app tasks; complete cross-app workflows requiring coordination across multiple mobile applications; achieve task success on a 150-task real-phone human evaluation and on AndroidWorld",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Cross-app workflows require completing a subtask in one app whose output/state must correctly transfer into a subsequent subtask in a different app, so a failure or state-tracking error in the first app breaks completion of the overall cross-app task; this is called out as the hardest, least-improved category.",
  "n_goals": "150-task human evaluation spanning apps, mini-apps, and cross-app workflows",
  "tracking_demand": "The agent must track UI/app state across a real, stateful phone environment (or the resettable PhoneWorld mock-app equivalent), including state that must persist and transfer correctly across multiple apps in cross-app workflows.",
  "scoring": "other:task-success-rate-by-category(app/mini-app/cross-app). Task success rate is reported by training stage (SFT/real-app RL/mixed RL), and gains are separately noted to be strongest on app/mini-app tasks while cross-app workflows remain an open challenge -- a per-category (subgoal-domain) breakdown rather than one aggregate score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "estimated-by-extractor",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Task success rate on the 150-task human evaluation improves from 36.67% (SFT) to 40.67% (real-app RL) to 45.33% (mixed RL); on AndroidWorld, from 60.3% to 77.2% to 83.2%; no external human/expert baseline beyond the human-graded evaluation itself.",
  "availability": null,
  "goal_span": "Across a 150-task human evaluation on real phones spanning apps, mini-apps, and cross-app workflows, task success rate improves from 36.67% after supervised fine-tuning to 40.67% after real-app RL and 45.33% after mixed RL... The gains are strongest on app and mini-app tasks, while long-horizontal cross-app workflows remain an important open challenge.",
  "horizon_span": "long-horizontal cross-app workflows remain an important open challenge",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289621498",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288670854",
  "title": "ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis",
  "year": 2026,
  "venue": null,
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 1,
  "publication_date": "2026-05-24",
  "months_since_pub": 4,
  "citations_per_month": 0.25,
  "artifact_name": "ScaleWoB",
  "artifact_kind": "environment/simulator",
  "domain": "web/GUI",
  "goal_types": "complete verifiable multi-step GUI tasks across mobile, desktop, or automotive/in-vehicle synthesized environments; succeed specifically on a distinguished long-horizon subset of tasks, which requires sustaining performance across more steps than the general task pool",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Tasks are generated with verifiable rewards over a fixed pipeline across many environments, and the paper explicitly separates a harder 'long-horizon subset' from the general task pool, implying these tasks chain more GUI steps together such that an early error compounds and depresses success more than on the general set (27.92% overall vs. 17.82% on the long-horizon subset).",
  "n_goals": "100+ environments and 1,000+ verifiable tasks overall; 120 challenging tasks across 63 simulated mobile apps in the released mobile GUI agent benchmark",
  "tracking_demand": "Agent must track GUI state across a chain of interface actions long enough to be classified into the 'long-horizon subset', where performance drops sharply relative to the general task pool, implying more state must be carried across more steps.",
  "scoring": "binary-final-success (per task), aggregated into a success-rate metric; the abstract reports success rate overall (27.92%) and separately for the long-horizon subset (17.82%), i.e. subset-level but not fine-grained subgoal-level credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Average success rate across five state-of-the-art mobile GUI agents is only 27.92%, dropping to 17.82% on the long-horizon subset, while humans reach 92.08%.",
  "availability": null,
  "goal_span": "Experiment results on five state-of-the-art mobile GUI agents reveal substantial headroom -- the average success rate is only 27.92\\%, dropping to 17.82\\% on long-horizon subset -- while humans reach 92.08\\%.",
  "horizon_span": "the average success rate is only 27.92\\%, dropping to 17.82\\% on long-horizon subset -- while humans reach 92.08\\%",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288670854",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "288977240",
  "title": "SentinelBench: A Benchmark for Long-Running Monitoring Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 2,
  "publication_date": "2026-06-03",
  "months_since_pub": 3,
  "citations_per_month": 0.67,
  "artifact_name": "SentinelBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "continuously monitor a live web environment (email/calendar/finance/professional-networking/entertainment) for a scripted external event; recognize the moment an event makes progress possible and act promptly; avoid excessive/wasteful actions (continuous polling/refreshing) while waiting",
  "goal_origin": "emitted-by-environment-over-time",
  "decomposition": "sequential-chain",
  "interdependence": "The agent's reaction is only valid once the scripted trigger event has actually occurred; acting too early wastes resources or fails the task, while acting too late incurs a reaction-time cost, so monitoring and action timing are tightly coupled to the environment's unfolding, injected event schedule.",
  "n_goals": "100 tasks across 10 synthetic web environments; each task built around a default 10-minute window (adjustable via a speed_factor)",
  "tracking_demand": "The agent must track whether the monitored page state has changed to reflect the awaited event, its own resource expenditure (tool calls/tokens) over the monitoring window, and elapsed time relative to the task's (possibly stretched) time budget.",
  "scoring": "other:multi-metric -- task completion, reaction time, and resource use are reported as separate metrics rather than a single pass/fail score, explicitly exposing the responsiveness-vs-cost tradeoff; no subgoal-checkpoint rubric beyond these three metrics is described.",
  "horizon_value": "each of the 100 tasks is designed to be achievable within a default 10-minute window; a speed_factor parameter (default 1.0) can stretch tasks to much longer durations (e.g. at speed_factor 0.25, tasks may require as long as 40 minutes)",
  "horizon_unit": "wall-clock-minutes",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "SentinelBench measures task completion, reaction time, and resource use, exposing the tradeoff between responsiveness and cost.",
  "horizon_span": "By default, each of the 100 tasks is designed to be achievable within 10 minutes ... a speed_factor parameter (default: 1.0) can be applied to stretch tasks to much longer durations ... tasks may require as long as 40 minutes to complete",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:288977240",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "emitted-by-environment-over-time",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "wall-clock-minutes",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "289097181",
  "title": "iOSWorld: A Benchmark for Personally Intelligent Phone Agents",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 3,
  "publication_date": "2026-06-08",
  "months_since_pub": 3,
  "citations_per_month": 1.0,
  "artifact_name": "iOSWorld",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete single-app tasks within one iOS app (27 tasks); complete multi-app task chains spanning 2 to 8 apps (60 tasks); infer personal patterns from persistent user data for memory/personalization tasks (46 tasks)",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Multi-app tasks require carrying data and state (transactions, messages, travel records, social relationships, financial activity) across up to 8 connected apps, and personalization tasks depend on correctly inferred patterns from that same persistent user identity.",
  "n_goals": "133 tasks total: 27 single-app, 60 multi-app (2-8 apps), 46 memory/personalization",
  "tracking_demand": "Agent must track a persistent user identity and its connected data (transactions, messages, travel records, social relationships, financial activity) across apps and infer behavioral patterns for personalization tasks.",
  "scoring": "other:success-rate-per-task-category",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "The best configuration reaches 52% overall but only 37% on multi-app tasks; privileged vision+XML access improves frontier models by up to 26 percentage points (no specific single model named for the top score; no human baseline reported).",
  "availability": null,
  "goal_span": "iOSWorld includes 133 tasks across three increasingly difficult categories. Single-app tasks (27) test one app, multi-app tasks (60) span 2 to 8 apps, and memory and personalization tasks (46) require agents to infer patterns from personal data.",
  "horizon_span": "multi-app tasks (60) span 2 to 8 apps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289097181",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "289690459",
  "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States",
  "year": 2026,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 3,
  "publication_date": "2026-06-30",
  "months_since_pub": 3,
  "citations_per_month": 1.0,
  "artifact_name": "unnamed mobile GUI benchmark (paper introduces the ATMem method)",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "act on every list entry that satisfies the given instruction (positive matches); reject/skip every list entry that violates the instruction's constraints (negative matches)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "None beyond a shared instruction: each entry's accept/reject decision is evaluated independently, though the agent must avoid missing entries or re-acting on already-handled ones across the list.",
  "n_goals": null,
  "tracking_demand": "The agent must track which near-identical list entries it has already acted on vs. still pending, and correctly apply the instruction's inclusion/exclusion constraints to each entry across a long trajectory.",
  "scoring": "other:custom-dual-metric -- App-Level Progress and Scope-Aware F1 separately measure completion of in-scope actions and avoidance of out-of-scope actions, a form of fine-grained (per-entry) credit rather than a single binary success label.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "From a list of near identical entries, agents must act on every entry that satisfies the instruction and reject entries that violate its constraints. We further introduce App-Level Progress and Scope-Aware F1 to measure these two dimensions separately.",
  "horizon_span": "Mobile GUI agents increasingly face long-horizon tasks that require reading, updating, and reusing task-relevant data across pages and applications.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:289690459",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279318971",
  "title": "Mirage-1: Augmenting and Updating GUI Agent with Hierarchical Multimodal Skills",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 13,
  "publication_date": "2025-06-12",
  "months_since_pub": 15,
  "citations_per_month": 0.87,
  "artifact_name": "AndroidLH (via Mirage-1)",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete real-world long-horizon, multi-app Android task scenarios using previously acquired hierarchical skills; correctly apply execution skills, core skills, and meta-skills at the appropriate level of abstraction across a long-horizon task",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Long-horizon, multi-app operations require correctly sequencing lower-level execution skills into higher-level core/meta-skills, so an error at the execution-skill level can derail the higher-level plan; the offline-to-online domain gap means offline-acquired skills must generalize without breaking this hierarchy.",
  "n_goals": "30 diverse tasks across multiple applications (AndroidLH)",
  "tracking_demand": "The agent must track which level of its hierarchical skill structure (execution, core, meta) is currently relevant, maintain state as it moves across multiple apps in a long-horizon scenario, and bridge the offline-to-online domain gap without losing track of previously acquired skills.",
  "scoring": "other:completion-and-success-rate -- Mirage-1 is reported to outperform previous agents by specific percentage-point margins (e.g. +79% on AndroidLH) on completion rate/success rate metrics; the abstract does not describe a fine-grained subgoal-checkpoint rubric beyond these benchmark-level metrics.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Mirage-1 outperforms previous agents by 32%, 19%, 15%, and 79% on AndroidWorld, MobileMiniWob++, Mind2Web-Live, and AndroidLH respectively; no human/expert baseline gap is reported in the abstract.",
  "availability": null,
  "goal_span": "To validate the performance of Mirage-1 in real-world long-horizon scenarios, we constructed a new benchmark, AndroidLH. Experimental results show that Mirage-1 outperforms previous agents by 32%, 19%, 15%, and 79% on AndroidWorld, MobileMiniWob++, Mind2Web-Live, and AndroidLH, respectively.",
  "horizon_span": "To validate the performance of Mirage-1 in real-world long-horizon scenarios, we constructed a new benchmark, AndroidLH.",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279318971",
  "provenance": "forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284153707",
  "title": "AndroidLens: Long-latency Evaluation with Nested Sub-targets for Android GUI Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 2,
  "publication_date": "2025-12-24",
  "months_since_pub": 9,
  "citations_per_month": 0.22,
  "artifact_name": "AndroidLens",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete nested sub-targets within a single long-latency mobile task; satisfy multi-constraint, multi-goal task requirements drawn from 38 real-world domains; make measurable milestone-level progress even when full task success is not reached",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Tasks are explicitly structured with nested sub-targets, and 'static evaluation... allows multiple valid paths', so sub-targets must be satisfied in some valid order while sharing device/app state, with milestone-based dynamic evaluation tracking incremental progress via Average Task Progress (ATP).",
  "n_goals": "571 tasks (average >26 steps each) across 38 domains",
  "tracking_demand": "Agent must track which nested sub-targets have been completed, tolerate environmental anomalies, and retain long-term memory of earlier steps across an average of more than 26 steps per task.",
  "scoring": "subgoal-checkpoint-partial-credit \u2014 explicitly, 'dynamic evaluation that employs a milestone-based scheme for fine-grained progress measurement via Average Task Progress (ATP)', separate from binary task success rate.",
  "horizon_value": ">26 (average, 'more than 26')",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Even the best models reach only a 12.7% task success rate and 50.47% Average Task Progress (ATP) (no human baseline given).",
  "availability": null,
  "goal_span": "(3) dynamic evaluation that employs a milestone-based scheme for fine-grained progress measurement via Average Task Progress (ATP)",
  "horizon_span": "comprising 571 long-latency tasks in both Chinese and English environments, each requiring an average of more than 26 steps to complete",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284153707",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "283439364",
  "title": "Benchmarking In-context Experiential Learning Through Repeated Product Recommendations",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain-other",
  "secondary_family": "personal-assistant-memory",
  "info_seeking_component": false,
  "domain_detail": "other:personalized-recommendation-experiential-learning",
  "citation_count": 2,
  "publication_date": "2025-11-27",
  "months_since_pub": 10,
  "citations_per_month": 0.2,
  "artifact_name": "BELA (Benchmark for Experiential Learning and Active exploration)",
  "artifact_kind": "benchmark",
  "domain": "other:personalized-recommendation-experiential-learning",
  "goal_types": "elicit unknown customer preferences through questions within a single recommendation interaction (episode); tailor questioning/recommendation strategy based on patterns observed across multiple prior episodes (customers/products)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Within an episode, later elicitation questions depend on what has already been learned about that customer's preferences in earlier turns; across episodes, the agent's within-episode strategy should improve based on the shared latent structure of customer/product distributions observed in prior episodes.",
  "n_goals": "benchmark built from a catalog of real-world Amazon products (71K products, 2K choice sets) and a diverse set of synthetic customer personas, evaluated over multiple interactions/episodes per agent",
  "tracking_demand": "The agent must track what it has learned about the current customer's preferences within an episode (turn-by-turn), and must also track/aggregate patterns across multiple prior episodes to adapt its questioning/recommendation strategy over time.",
  "scoring": "continuous-reward -- the benchmark measures whether agents can exploit consistent latent preferences across episodes, finding that models can learn across turns, but struggle to improve across episodes; a continuous, comparative performance measure rather than a discrete subgoal-checkpoint rubric.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "We instantiate the Benchmark for Experiential Learning and Active exploration (BELA) by combining (1) a rich catalog of real-world products from Amazon, (2) a diverse collection of synthetic customer personas aimed to capture heterogeneous latent preferences, and (3) an LLM-based customer simulator framework that emulate preference-revealing interactions... Benchmarking current models reveals that they can learn across turns, but struggle to improve across episodes.",
  "horizon_span": "we instantiate the Benchmark for Experiential Learning and Active exploration (BELA) by combining (1) a rich catalog of real-world products from Amazon, (2) a diverse collection of synthetic customer personas",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283439364",
  "provenance": "forward-citation",
  "scoring_family": "continuous-reward",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "276574720",
  "title": "MobileSteward: Integrating Multiple App-Oriented Agents with Self-Evolution to Automate Cross-App Instructions",
  "year": 2025,
  "venue": "Knowledge Discovery and Data Mining",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 24,
  "publication_date": "2025-02-24",
  "months_since_pub": 19,
  "citations_per_month": 1.26,
  "artifact_name": "CAPBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "associate and sequence sub-tasks across multiple mobile apps per a cross-app instruction; assign each associated sub-task to the correct app-oriented StaffAgent; avoid error propagation and information loss across the multi-step, multi-app execution; complete each of the 500 cross-app instructions spanning 14 apps in 6 categories",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Dynamic Recruitment builds a scheduling graph guided by information flow that explicitly associates tasks among apps, so a StaffAgent's sub-task execution depends on information flowing correctly from another app-oriented StaffAgent's prior output; Adjusted Evaluation exists specifically to catch error propagation/information loss between these dependent steps.",
  "n_goals": "500 cross-app instructions across 14 apps in 6 categories",
  "tracking_demand": "The centralized StewardAgent must track the scheduling graph of inter-app task associations, information flow between StaffAgents, and self-evolving memory of past executions to avoid repeating errors across a cross-app instruction.",
  "scoring": "other:step-rate-and-success-rate",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "MobileSteward achieves the best performance compared to both single-agent and multi-agent frameworks on CAPBench; no single numeric headline figure or human baseline given in the abstract.",
  "availability": null,
  "goal_span": "we propose a self-evolving multi-agent framework named MobileSteward which integrates multiple app-oriented StaffAgents coordinated by a centralized StewardAgent. ... Dynamic Recruitment generates a scheduling graph guided by information flow to explicitly associate tasks among apps. ... We establish the first English Cross-APP Benchmark (CAPBench) in the real-world environment",
  "horizon_span": "Step Rate measures the accuracy of individual actions executed by the StaffAgents.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:276574720",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "282139005",
  "title": "ColorBench: Benchmarking Mobile Agents with Graph-Structured Framework for Complex Long-Horizon Tasks",
  "year": 2025,
  "venue": "The Web Conference",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 8,
  "publication_date": "2025-10-16",
  "months_since_pub": 11,
  "citations_per_month": 0.73,
  "artifact_name": "ColorBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete a single-app mobile task via one of multiple valid GUI action paths; complete a cross-app mobile task requiring coordination across multiple applications; reach subtask-level completion milestones within a longer task",
  "goal_origin": "given-up-front",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Tasks are represented as a graph of finite states, with 'at least two correct paths and several typical error paths' per task, so subtasks can often be completed via alternative orderings/routes while still sharing an overall precedence structure toward task completion.",
  "n_goals": "175 tasks (74 single-app, 101 cross-app), average length >13 steps",
  "tracking_demand": "Agent must track which subtasks have been completed, which of several valid paths it is following through the task's state graph, and avoid known error paths, across an average of more than 13 steps.",
  "scoring": "subgoal-checkpoint-partial-credit \u2014 explicitly, ColorBench 'supports evaluation of multiple valid solutions, subtask completion rate statistics, and atomic-level capability analysis', i.e. subtask-level credit distinct from a single golden-path pass/fail.",
  "horizon_value": ">13 (average, 'over 13 steps')",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": null,
  "availability": null,
  "goal_span": "we develop ColorBench, a benchmark focused on complex long-horizon tasks. It supports evaluation of multiple valid solutions, subtask completion rate statistics, and atomic-level capability analysis.",
  "horizon_span": "ColorBench contains 175 tasks (74 single-app, 101 cross-app) with an average length of over 13 steps",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:282139005",
  "provenance": "forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279118560",
  "title": "DeepShop: A Benchmark for Deep Research Shopping Agents",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 42,
  "publication_date": "2025-06-03",
  "months_since_pub": 15,
  "citations_per_month": 2.8,
  "artifact_name": "DeepShop",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "satisfy multiple product-attribute constraints in a single shopping query; apply the correct search filters specified or implied by the query; apply the correct sorting preference specified or implied by the query; achieve overall shopping-task success across easy/medium/hard complexity tiers",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "Product attributes, search filters, and sorting preferences must all be satisfied simultaneously within the same search/browse session, so applying one filter narrows the candidate set relevant to satisfying the others, and query complexity increases as more of these are combined via successive 'evolutions'.",
  "n_goals": "queries evolved to three complexity levels (easy/medium/hard) based on the number of evolutions combining product attributes, filters, and sorting preferences",
  "tracking_demand": "Agent must track which product attributes, filters, and sorting preferences the query requires, and verify each is correctly reflected in its final shopping actions/results.",
  "scoring": "subgoal-checkpoint-partial-credit - proposes a fine-grained and holistic evaluation framework assessing agent performance on individual aspects (product attributes, search filters, sorting preferences) as well as an overall success rate; an explicit per-aspect (subgoal-level) partial-credit scheme in addition to the holistic score.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "RAG struggles with complex queries due to lack of web interaction, while other methods face significant challenges with filters and sorting preferences, leading to low overall success rates; no specific numeric top score or human baseline is given.",
  "availability": null,
  "goal_span": "(3) Fine-grained and holistic evaluation: We propose an automated evaluation framework that assesses agent performance in terms of fine-grained aspects (product attributes, search filters, and sorting preferences) and reports the overall success rate through holistic evaluation.",
  "horizon_span": "We further evolve these queries to increase complexity, considering product attributes, search filters, and sorting preferences, and classify them into three levels: easy, medium, and hard, based on the number of evolutions.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279118560",
  "provenance": "asta-find,parametric",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280635680",
  "title": "MVISU-Bench: Benchmarking Mobile Agents for Real-World Tasks by Multi-App, Vague, Interactive, Single-App and Unethical Instructions",
  "year": 2025,
  "venue": "ACM Multimedia",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 7,
  "publication_date": "2025-08-12",
  "months_since_pub": 13,
  "citations_per_month": 0.54,
  "artifact_name": "MVISU-Bench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete multi-app instructions requiring cross-app subgoal coordination; clarify vague/underspecified user instructions before acting; handle interactive instructions requiring mid-task clarification; complete single-app instructions; recognize and appropriately refuse or handle unethical instructions",
  "goal_origin": "mixed:given-up-front-with-interactive-clarification-injected-mid-episode",
  "decomposition": "other:five-independent-instruction-categories-with-multi-app-subset-requiring-sequential-cross-app-steps",
  "interdependence": "Within Multi-App tasks, steps completed in one application constrain or set up state needed in a subsequent application; the five instruction categories are otherwise independent test conditions.",
  "n_goals": "404 tasks across 137 mobile applications",
  "tracking_demand": "The agent must track task state across multiple mobile apps for Multi-App instructions, detect ambiguity requiring clarification for Vague/Interactive instructions, and recognize when an instruction should be refused for Unethical instructions.",
  "scoring": "other:binary-success-rate-per-instruction-category",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Aider (the proposed prompting module) improves overall success rate by 19.55% over the prior SOTA on MVISU-Bench, with 53.52% and 29.41% gains on unethical and interactive instructions respectively; no human/expert ceiling is reported.",
  "availability": null,
  "goal_span": "we present MVISU-Bench, a bilingual benchmark that includes 404 tasks across 137 mobile applications",
  "horizon_span": "we present MVISU-Bench, a bilingual benchmark that includes 404 tasks across 137 mobile applications",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280635680",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "mixed",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "283072838",
  "title": "Mobile-Agent-RAG: Driving Smart Multi-Agent Coordination with Contextual Knowledge Empowerment for Long-Horizon Mobile Automation",
  "year": 2025,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "multi-agent-org",
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 2,
  "publication_date": "2025-11-15",
  "months_since_pub": 10,
  "citations_per_month": 0.2,
  "artifact_name": "Mobile-Eval-RAG",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete a cross-application mobile-automation task requiring both high-level plan steps and precise low-level UI operations; correctly execute app-specific atomic UI actions aligned with the current subtask",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "High-level plan steps must be correctly grounded into low-level, app-specific UI operations; an error at the planning level (strategic hallucination) or execution level (operational error on a specific app's UI) breaks the chain needed to complete a multi-app task.",
  "n_goals": null,
  "tracking_demand": "Agent must track which app/subtask it is currently operating in, the current step of the high-level plan, and precise UI state needed for accurate atomic actions across multiple apps in one task.",
  "scoring": "other:task-completion-rate-plus-step-efficiency - reports task completion rate and step efficiency (Mobile-Agent-RAG improves task completion rate by 11.0% and step efficiency by 10.2%); abstract does not describe an explicit subgoal-checkpoint partial-credit scheme beyond these two metrics.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not applicable - no horizon figure is given",
  "horizon_stated": "no",
  "headline_result": "Mobile-Agent-RAG improves task completion rate by 11.0% and step efficiency by 10.2% over SoTA baselines on Mobile-Eval-RAG; no human/expert baseline is reported.",
  "availability": null,
  "goal_span": "Furthermore, we introduce Mobile-Eval-RAG, a challenging benchmark for evaluating such agents on realistic multi-app, long-horizon tasks.",
  "horizon_span": "Mobile agents show immense potential, yet current state-of-the-art (SoTA) agents exhibit inadequate success rates on real-world, long-horizon, cross-application tasks.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283072838",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "284078735",
  "title": "MobileWorld: Benchmarking Autonomous Mobile Agents in Agent-User Interactive and MCP-Augmented Environments",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 58,
  "publication_date": "2025-12-22",
  "months_since_pub": 9,
  "citations_per_month": 6.44,
  "artifact_name": "MobileWorld",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete long-horizon, cross-application mobile workflows spanning up to 20 applications; handle vague user instructions and hybrid tool usage; coordinate agent-user interaction and MCP-augmented tool calls mid-task",
  "goal_origin": "mixed:given-up-front-with-injected-user-interaction-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Multi-app tasks (62.2% of the benchmark) require carrying state across applications and coordinating with MCP tool calls and agent-user interaction, so earlier steps constrain what later cross-app or interactive steps must accomplish.",
  "n_goals": "201 tasks across 20 applications; 62.2% are multi-app tasks (vs. 9.5% in AndroidWorld)",
  "tracking_demand": "Agent must track cross-application state across nearly twice as many completion steps on average (27.8 vs. 14.3) as AndroidWorld, while also handling user-interaction requests and MCP-tool-call state.",
  "scoring": "binary-final-success",
  "horizon_value": "27.8 (vs. 14.3 in AndroidWorld)",
  "horizon_unit": "agent-steps",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task",
  "horizon_stated": "yes",
  "headline_result": "Best agentic framework and end-to-end model achieve 51.7% and 20.9% success rates respectively (vs. AndroidWorld's >90% saturation for prior agents); no human baseline reported.",
  "availability": null,
  "goal_span": "We introduce MobileWorld, a substantially more challenging benchmark designed to reflect real-world usage through 201 tasks across 20 applications. MobileWorld derives its difficulty from an emphasis on long-horizon, cross-application workflows, requiring nearly twice as many completion steps on average (27.8 vs. 14.3) and featuring a significantly higher proportion of multi-app tasks (62.2% vs. 9.5%) than AndroidWorld.",
  "horizon_span": "requiring nearly twice as many completion steps on average (27.8 vs. 14.3) and featuring a significantly higher proportion of multi-app tasks (62.2% vs. 9.5%) than AndroidWorld",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:284078735",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "mixed",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "agent-steps",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "287510062",
  "title": "NaturalGAIA: A Verifiable Benchmark and Hierarchical Framework for Long-Horizon GUI Tasks",
  "year": 2025,
  "venue": "Annual Meeting of the Association for Computational Linguistics",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 0,
  "publication_date": "2025-08-02",
  "months_since_pub": 13,
  "citations_per_month": 0.0,
  "artifact_name": "NaturalGAIA",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "decompose a natural, non-linear human GUI intent into a structured Task Topology of atomic sub-tasks; dynamically schedule sub-tasks across heterogeneous agents via context evolution; execute each atomic sub-task with precision via hybrid visual-structural perception; achieve a high Weighted Pathway Success Rate across the full causal pathway",
  "goal_origin": "self-generated-by-agent",
  "decomposition": "DAG-with-precedence",
  "interdependence": "Human GUI intents are characterized by cognitive non-linearity and contextual dependencies, and the framework's Context Evolution mechanism exists specifically to bridge information gaps between steps, meaning later atomic sub-tasks depend on state produced by earlier ones within the decomposed Task Topology.",
  "n_goals": null,
  "tracking_demand": "The manager must dynamically track the Task Topology of atomic sub-tasks, schedule them across heterogeneous agents, and evolve shared context to bridge information gaps between dependent steps, assessed via a hierarchical success/error-attribution framework.",
  "scoring": "other:weighted-pathway-success-rate",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "LightManus-Jarvis achieves a Weighted Pathway Success Rate of 45.6% vs. the SOTA baseline's 21.1%, while reducing token consumption by 75% and execution time by 76%; no human baseline given.",
  "availability": null,
  "goal_span": "By decoupling logical causal pathways from linguistic narratives, it rigorously simulates natural human intent, characterized by cognitive non-linearity and contextual dependencies. ... The parser decomposes abstract user intents into a structured Task Topology composed of atomic tasks. ... Experiments demonstrate that our approach achieves a Weighted Pathway Success Rate of 45.6%, significantly outperforming the state-of-the-art baseline (21.1%)",
  "horizon_span": "Experiments demonstrate that our approach achieves a Weighted Pathway Success Rate of 45.6%, significantly outperforming the state-of-the-art baseline (21.1%), while reducing token consumption by 75% and execution time by 76%.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:287510062",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "self-generated-by-agent",
  "decomposition_family": "DAG-with-precedence",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "277780301",
  "title": "RealWebAssist: A Benchmark for Long-Horizon Web Assistance with Real-World Users",
  "year": 2025,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "in",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 26,
  "publication_date": "2025-04-14",
  "months_since_pub": 17,
  "citations_per_month": 1.53,
  "artifact_name": "RealWebAssist",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "correctly follow each instruction in a sequence of real, sequentially-issued user instructions across multiple websites; reason about the true intent behind ambiguous instructions; keep track of the user's mental state and user-specific routines as instructions evolve over the session; ground each intended task to the correct GUI element on the current website",
  "goal_origin": "injected-by-user-mid-episode",
  "decomposition": "sequential-chain",
  "interdependence": "Because instructions are sequential and 'may evolve over time, reflecting changes in the user's mental state', later instructions can only be correctly interpreted in light of the user's accumulated mental-state/routine context built up from earlier instructions in the same session.",
  "n_goals": null,
  "tracking_demand": "Agent must retain a running model of the user's intent, mental state, and personal routines across a long sequence of instructions issued over multiple websites, using this history to correctly interpret each new (sometimes ambiguous) instruction.",
  "scoring": "other:not-stated precisely \u2014 the abstract reports that 'state-of-the-art models struggle to understand and ground user instructions' but does not describe an explicit partial-credit/checkpoint scoring scheme across the instruction sequence.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "RealWebAssist includes a dataset of sequential instructions collected from real-world human users. Each user instructs a web-based assistant to perform a series of tasks on multiple websites.",
  "horizon_span": "To achieve successful assistance with long-horizon web-based tasks, AI agents must be able to sequentially follow real-world user instructions over a long period.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:277780301",
  "provenance": "asta-find",
  "scoring_family": "not recorded",
  "goal_origin_family": "injected-by-user-mid-episode",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "280536823",
  "title": "ShoppingBench: A Real-World Intent-Grounded Shopping Benchmark for LLM-based Agents",
  "year": 2025,
  "venue": "AAAI Conference on Artificial Intelligence",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 27,
  "publication_date": "2025-08-06",
  "months_since_pub": 13,
  "citations_per_month": 2.08,
  "artifact_name": "ShoppingBench",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "apply vouchers correctly to a purchase; manage a budget across a shopping session; find and select from multi-product sellers matching a grounded intent; satisfy increasingly challenging levels of grounded shopping intent end-to-end",
  "goal_origin": "given-up-front",
  "decomposition": "other:joint-constraint-satisfaction-within-one-shopping-task",
  "interdependence": "Voucher application, budget limits, and multi-product-seller selection must all be jointly satisfied within a single shopping session, since applying a voucher changes the effective budget available for remaining choices.",
  "n_goals": null,
  "tracking_demand": "Agent must track running spend/budget, applied vouchers, and multi-seller product-matching requirements across a session, operating within a sandbox of over 2.5 million real-world products.",
  "scoring": "binary-final-success",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even GPT-4.1 achieves an absolute success rate under 50% on ShoppingBench tasks; a distilled smaller agent trained via SFT+RL on synthetic trajectories reaches competitive performance versus GPT-4.1 (no human baseline reported).",
  "availability": null,
  "goal_span": "real-world users often pursue more complex goals, such as applying vouchers, managing budgets, and finding multi-products seller. To bridge this gap, we propose ShoppingBench, a novel end-to-end shopping benchmark designed to encompass increasingly challenging levels of grounded intent.",
  "horizon_span": "we provide a large-scale shopping sandbox that serves as an interactive simulated environment, incorporating over 2.5 million real-world products",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280536823",
  "provenance": "asta-find",
  "scoring_family": "binary-final-success",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "other (singleton)",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "binary final success only",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "283438874",
  "title": "ShoppingComp: Are LLMs Really Ready for Your Shopping Cart?",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 8,
  "publication_date": "2025-11-28",
  "months_since_pub": 10,
  "citations_per_month": 0.8,
  "artifact_name": "ShoppingComp",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "retrieve products satisfying many simultaneous discovery constraints; generate an expert-level report on the retrieved products; make a safety-critical purchase decision (e.g. flag unsafe product usage)",
  "goal_origin": "given-up-front",
  "decomposition": "set-of-independent",
  "interdependence": "The three capabilities (retrieval, report generation, safety decision) are evaluated together per instance, and an agent's retrieval errors propagate into report inaccuracies and unsafe recommendations, so the three sub-goals are coupled through the shared product-cart context.",
  "n_goals": "145 instances and 558 scenarios (curated by 35 experts)",
  "tracking_demand": "The agent must track which of the multiple product-discovery constraints have been satisfied, the evidence supporting its report, and any identified safety hazards, while operating in an open-world product catalogue.",
  "scoring": "other:accuracy-percentage-per-model",
  "horizon_value": null,
  "horizon_unit": "other:not-stated",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "not stated",
  "horizon_stated": "no",
  "headline_result": "Even state-of-the-art models achieve low performance (e.g. 17.76% for GPT-5.2, 15.82% for Gemini-3-Pro); no human baseline given.",
  "availability": "https://github.com/ByteDance-BandAI/ShoppingComp",
  "goal_span": "ShoppingComp, a challenging real-world benchmark for comprehensively evaluating LLM-powered shopping agents on three core capabilities: precise product retrieval, expert-level report generation, and safety critical decision making. ... The benchmark comprises 145 instances and 558 scenarios",
  "horizon_span": "The benchmark comprises 145 instances and 558 scenarios, curated by 35 experts to reflect authentic shopping needs.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:283438874",
  "provenance": "asta-find",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "set-of-independent",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "other (unclassified)",
  "subgoal_credit": "no"
 },
 {
  "corpusId": "279261251",
  "title": "Atomic-to-Compositional Generalization for Mobile Agents with A New Benchmark and Scheduling System",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 12,
  "publication_date": "2025-06-10",
  "months_since_pub": 15,
  "citations_per_month": 0.8,
  "artifact_name": "UI-NEXUS",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "complete compositional mobile operations that concatenate multiple atomic tasks (Simple Concatenation); transition context correctly across sub-tasks within one compositional task (Context Transition); perform a deep multi-step drill-down within an app or workflow (Deep Dive)",
  "goal_origin": "given-up-front",
  "decomposition": "hierarchical",
  "interdependence": "Compositional tasks require correctly chaining atomic subtasks in sequence, carrying context across sub-task boundaries, and drilling deeper within a sub-task, so failure to track completed atomic steps and context causes cascading errors (under-execution, over-execution, attention drift).",
  "n_goals": "100 interactive task templates, average optimal step count 14.05",
  "tracking_demand": "The agent must track which atomic subtasks within a compositional task it has already completed, the context state carried between apps/screens, and overall progress toward each of the three compositional operation types.",
  "scoring": "other:task-success-rate-with-efficiency. The abstract reports task success rate together with efficiency (step count vs. optimal), not a subgoal-checkpoint credit scheme; no partial credit for completed atomic subtasks within a compositional task is described.",
  "horizon_value": "average optimal step count of 14.05 (over 100 interactive task templates)",
  "horizon_unit": "actions",
  "horizon_basis": "stated-by-authors",
  "horizon_scope": "per task (average across the 100 interactive task templates)",
  "horizon_stated": "yes",
  "headline_result": "AGENT-NEXUS achieves a 24% to 40% task success rate improvement for existing mobile agents on compositional operation tasks; no explicit human/expert comparison given in the abstract.",
  "availability": "https://ui-nexus.github.io",
  "goal_span": "UI-NEXUS supports interactive evaluation in 20 fully controllable local utility app environments, as well as 30 online Chinese and English service apps. It comprises 100 interactive task templates with an average optimal step count of 14.05.",
  "horizon_span": "It comprises 100 interactive task templates with an average optimal step count of 14.05.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279261251",
  "provenance": "asta-find,forward-citation,web-registry",
  "scoring_family": "other (singleton)",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "hierarchical",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "280536268",
  "title": "VeriWeb: Verifiable Long-Chain Web Benchmark for Agentic Information-Seeking",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 10,
  "publication_date": "2025-08-06",
  "months_since_pub": 13,
  "citations_per_month": 0.77,
  "artifact_name": "VeriWeb",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "ensure comprehensive information coverage across breadth- and depth-oriented multi-hop web searches; complete a sequence of interdependent, individually-verifiable subtasks within one long-chain web task; maintain consistent context tracking across a long information-seeking chain",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Tasks are decomposed into a sequence of interdependent, verifiable subtasks, so each subtask-level answer must remain correct and unchanged for later subtasks (and the final answer) to be verifiable and correct.",
  "n_goals": "302 tasks across five real-world domains",
  "tracking_demand": "Agent must track and verify each subtask-level answer as it progresses through a long chain of interdependent web-search subtasks, ensuring comprehensive information coverage and consistent context tracking across the chain.",
  "scoring": "subgoal-checkpoint-partial-credit",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": null,
  "availability": null,
  "goal_span": "(2) subtask-level verifiability, where tasks are decomposed into a sequence of interdependent verifiable subtasks. This structure enables diverse exploration strategies within each subtask, while ensuring that each subtask-level answer remains unchanged and verifiable.",
  "horizon_span": "The benchmark consists of 302 tasks across five real-world domains, each with a complete trajectory demonstration, annotated by human experts.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:280536268",
  "provenance": "asta-find,forward-citation",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 },
 {
  "corpusId": "279119756",
  "title": "WebChoreArena: Evaluating Web Browsing Agents on Realistic Tedious Web Tasks",
  "year": 2025,
  "venue": "arXiv.org",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": "information-seeking",
  "info_seeking_component": true,
  "domain_detail": "web/GUI",
  "citation_count": 28,
  "publication_date": "2025-06-02",
  "months_since_pub": 15,
  "citations_per_month": 1.87,
  "artifact_name": "WebChoreArena",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "retain and retrieve large amounts of information gathered from many observations (Massive Memory); perform precise mathematical/quantitative reasoning over collected information (Calculation); keep information consistent while tracking it across multiple webpages over the course of a task (Long-Term Memory)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Later calculation and consistency-checking steps depend on information correctly retained from earlier observations across multiple webpages within the same reproducible WebArena environments.",
  "n_goals": "532 tasks",
  "tracking_demand": "Agent must accumulate and correctly recall large amounts of information across many webpages, perform arithmetic over it, and keep facts consistent across the whole task rather than a single page view.",
  "scoring": "other:not-stated \u2014 the abstract reports overall performance improvements with LLM scale and a persisting gap versus WebArena, but does not describe subgoal-level or partial credit within a task.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Even GPT-5 leaves 'substantial room for improvement compared to WebArena'; no specific numeric score or human baseline given in the abstract.",
  "availability": null,
  "goal_span": "It systematically expands the evaluation space along three critical dimensions: (i) $\\textbf{Massive Memory}$... (ii) $\\textbf{Calculation}$... and (iii) $\\textbf{Long-Term Memory}$, necessitating consistent information tracking across multiple webpages.",
  "horizon_span": "WebChoreArena introduces 532 carefully curated tasks developed over 300+ hours",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:279119756",
  "provenance": "asta-find,parametric",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "280676456",
  "title": "WebMall - A Multi-Shop Benchmark for Evaluating Web Agents",
  "year": 2025,
  "venue": null,
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 14,
  "publication_date": "2025-08-18",
  "months_since_pub": 13,
  "citations_per_month": 1.08,
  "artifact_name": "WebMall",
  "artifact_kind": "benchmark",
  "domain": "web/GUI",
  "goal_types": "find a specific product across four simulated online shops; perform price comparisons across shops; identify suitable substitutes or compatible products (advanced search); add items to a cart and complete checkout in the correct shop(s)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Cross-shop comparison tasks (price comparison, substitute-finding) require the agent to have gathered and correctly retained product information from multiple shops before it can complete a later checkout/decision step, so earlier per-shop searches constrain what a correct final answer/purchase can be.",
  "n_goals": "91 cross-shop tasks across four simulated shops (per web search results describing the arXiv paper; not independently confirmed against the original abstract, which was unavailable in this row)",
  "tracking_demand": "Agent must retain product/price information gathered from each of the four shops it has visited so far in order to correctly compare, substitute, or complete a checkout later in the same task.",
  "scoring": "other:not-stated precisely \u2014 reported as task completion rates by category (e.g. below 65% for cheapest-product and vague-product search among top agents), which is closer to binary-final-success per task than explicit subgoal partial credit.",
  "horizon_value": null,
  "horizon_unit": null,
  "horizon_basis": null,
  "horizon_scope": null,
  "horizon_stated": "no",
  "headline_result": "Top-performing agents (among eight tested configurations) achieve task completion rates below 65% in the cheapest-product-search and vague-product-search categories (no human baseline given).",
  "availability": null,
  "goal_span": "LLM-based web agents have the potential to automate long-running web tasks, such as searching for products in multiple e-shops and subsequently ordering the cheapest products that meet the users needs.",
  "horizon_span": "the best-performing agents achieved task completion rates below 65% in the task categories cheapest product search and vague product search",
  "extraction_confidence": 1,
  "fit": "rephrased",
  "scope_flag": "Shard row for this paper had no abstract; goal-structure/horizon claims rely on text retrieved via web search/fetch rather than the packet's own input row, and the exact task count (91) comes from a secondary search-result summary, not a directly quoted original sentence \u2014 worth re-verifying against the source PDF if this paper is retained.",
  "url": "https://api.semanticscholar.org/CorpusId:280676456",
  "provenance": "parametric",
  "scoring_family": "not recorded",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "not recorded",
  "scoring_kind": "not recorded",
  "subgoal_credit": "unclear"
 },
 {
  "corpusId": "278782564",
  "title": "Web-Shepherd: Advancing PRMs for Reinforcing Web Agents",
  "year": 2025,
  "venue": "Neural Information Processing Systems",
  "tier": "relevant",
  "family": "web-GUI",
  "family_source": "extracted-domain",
  "secondary_family": null,
  "info_seeking_component": false,
  "domain_detail": "web/GUI",
  "citation_count": 29,
  "publication_date": "2025-05-21",
  "months_since_pub": 16,
  "citations_per_month": 1.81,
  "artifact_name": "WebPRM Collection / WebRewardBench",
  "artifact_kind": "dataset",
  "domain": "web/GUI",
  "goal_types": "navigate a website across many sequential steps toward a stated goal; at each step, satisfy sub-goals encoded in an annotated checklist (process-level correctness); complete the overall web-navigation task successfully (episode-level correctness)",
  "goal_origin": "given-up-front",
  "decomposition": "sequential-chain",
  "interdependence": "Each step in a web-navigation trajectory is checked against a checklist of expected sub-goals, and correct completion of the overall task depends on getting each of these ordered sub-steps right in sequence.",
  "n_goals": "40K step-level preference pairs with annotated checklists spanning diverse domains and difficulty levels",
  "tracking_demand": "The reward model must track which checklist sub-goals have already been satisfied at each step of a trajectory so it can assess the process-level (not just outcome-level) quality of a web-navigation trajectory.",
  "scoring": "subgoal-checkpoint-partial-credit. Web-Shepherd is explicitly built as a process reward model (PRM) that assesses web navigation trajectories 'in a step-level' via annotated checklists -- i.e. its purpose is to provide step/subgoal-level credit rather than only end-of-episode success.",
  "horizon_value": "trajectory lengths vary by difficulty: median ~5 steps (easy), ~9 steps (medium), ~20 steps (hard), with some trajectories exceeding 40 steps",
  "horizon_unit": "actions",
  "horizon_basis": "derived-from-a-reported-statistic",
  "horizon_scope": "per task/trajectory (median step counts differ by task difficulty tier)",
  "horizon_stated": "yes",
  "headline_result": "Web-Shepherd achieves about 30 points better accuracy than GPT-4o on WebRewardBench, and using Web-Shepherd as verifier with GPT-4o-mini as policy improves WebArena-lite performance by 10.9 points at 10x lower cost than using GPT-4o-mini as verifier; no human/expert baseline given.",
  "availability": null,
  "goal_span": "we propose the first process reward model (PRM) called Web-Shepherd which could assess web navigation trajectories in a step-level. To achieve this, we first construct the WebPRM Collection, a large-scale dataset with 40K step-level preference pairs and annotated checklists spanning diverse domains and difficulty levels.",
  "horizon_span": "Easy tasks generally require fewer steps (median \u2248 5), whereas medium tasks show more variability (median \u2248 9), and hard tasks involve significantly longer trajectories (median \u2248 20), with some exceeding 40 steps.",
  "extraction_confidence": 2,
  "fit": "explicit",
  "scope_flag": null,
  "url": "https://api.semanticscholar.org/CorpusId:278782564",
  "provenance": "asta-find",
  "scoring_family": "subgoal-checkpoint-partial-credit",
  "goal_origin_family": "given-up-front",
  "decomposition_family": "sequential-chain",
  "horizon_unit_family": "actions",
  "scoring_kind": "subgoal / checkpoint partial credit",
  "subgoal_credit": "yes"
 }
]