{
  "catalog_version": "1.0.0",
  "catalog_id": "greatruth-ai-data-index",
  "name": "GreaTruth AI Data Index",
  "description": "Alvance GreaTruth 后训练数据目录，覆盖可执行 RL 环境（RLVR）、真实与合成 SFT 轨迹、评测集与 Rubric，提供可运行样例、独立 Verifier、Schema 和下载归档。",
  "canonical_url": "https://display.greatruth.cloud/",
  "updated_at": "2026-09-29",
  "generated_at": "2026-09-28T18:47:16.289Z",
  "publisher": {
    "name": "Alvance",
    "type": "AI data company"
  },
  "product": {
    "name": "GreaTruth",
    "type": "AI data product"
  },
  "access": {
    "tier": "public_evaluation_sample",
    "authentication": "none",
    "terms": "/downloads/README.md",
    "note": "公开下载仅用于技术评估；完整数据与生产授权以 Alvance 项目协议为准。",
    "note_en": "Public downloads are provided for technical evaluation. Complete datasets and production licensing are governed by the applicable Alvance project agreement."
  },
  "product_lines": [
    {
      "id": "vertical_domain",
      "slug": "vertical-domain",
      "code": "PRODUCT 01",
      "title": "垂域数据",
      "title_en": "Vertical Domain Data",
      "eyebrow": "DOMAIN / PRACTICE",
      "description": "面向编程、咨询、产品经理、物流、电商、网安和通用办公场景的垂域任务样例，保留完整题面、输入材料、参考交付物、Rubric、评测记录和配套说明书。",
      "description_en": "Domain-specific task samples for coding, consulting, product management, logistics, e-commerce, cybersecurity, and general office work, retaining complete instructions, inputs, reference deliverables, rubrics, evaluation records, and companion guides.",
      "meta_description_en": "GreaTruth vertical-domain samples for coding, consulting, product, logistics, e-commerce, cybersecurity, and office tasks",
      "page_description_en": "This product line organizes task samples by industry and role. Each sample presents its instruction, environment materials, deliverables, evaluation boundaries, and guide; scores are not combined across samples.",
      "summary": "行业与岗位任务样例",
      "summary_en": "Industry and role task samples",
      "url": "/lines/vertical-domain/",
      "accent": "orange",
      "highlights": [
        "编程与形式化",
        "咨询与经营",
        "产品与物流",
        "电商与网安",
        "完整说明书"
      ],
      "highlights_en": [
        "Coding and formal methods",
        "Consulting and operations",
        "Product and logistics",
        "E-commerce and cybersecurity",
        "Complete guides"
      ],
      "proof_metrics": [
        {
          "label": "样例数量",
          "label_en": "Sample count",
          "value": "9",
          "hint": "7 个领域",
          "hint_en": "7 domains"
        },
        {
          "label": "完整题面",
          "label_en": "Complete instructions",
          "value": "9/9",
          "hint": "来源归档",
          "hint_en": "Source archives"
        },
        {
          "label": "说明书",
          "label_en": "Companion guides",
          "value": "9/9",
          "hint": "逐样例提供",
          "hint_en": "Per sample"
        }
      ],
      "dataset_ids": [
        "gr-vertical-coding-cl-t8-003-lean",
        "gr-vertical-coding-vhdl-t8-001-fifo",
        "gr-vertical-pm-t8-004",
        "gr-vertical-cons-t4-019",
        "gr-vertical-hv3-cons-0002",
        "gr-vertical-log-t8-l4",
        "gr-vertical-ecom-t1-001-gmv",
        "gr-vertical-cy-t8-001",
        "gr-vertical-hv3-gene-0001"
      ],
      "dataset_count": 9,
      "download": {
        "filename": "greatruth-vertical-domain-all-samples.zip",
        "url": "/downloads/greatruth-vertical-domain-all-samples.zip",
        "media_type": "application/zip",
        "byte_size": 272162538,
        "sha256": "a86bbddb02b24280de83fb8d5e527061a7dbcb75e6ffe477c56951fc8060c2bc",
        "dataset_count": 9,
        "file_count": 27
      }
    },
    {
      "id": "rl_sft_environment",
      "slug": "rl-sft",
      "code": "PRODUCT 02",
      "title": "RL强化训练数据",
      "title_en": "RL Training Data",
      "eyebrow": "REINFORCEMENT LEARNING / FEEDBACK",
      "description": "面向终端执行、软件工程、客户支持、专业办公与桌面操作的强化学习任务数据。样例按来源保留任务题面、环境或工作区、工具接口、Verifier 和评测记录，可用于核验训练任务及反馈结构。",
      "description_en": "Reinforcement learning task data for terminal execution, software engineering, customer support, professional office work, and desktop interaction. Samples retain source instructions, environments or workspaces, tool interfaces, verifiers, and evaluation records for inspecting tasks and feedback structures.",
      "meta_description_en": "GreaTruth RL training data: task samples for TerminalBench 4, DeepSWE, AutomationBench, JobBench, Computer Use, and ClawBench",
      "page_description_en": "Public samples cover terminal tasks, repository changes, customer-support coordination, document and spreadsheet delivery, and desktop and browser actions. Each sample shows its supplied instructions, environment materials, verification methods, and run records. Missing materials are identified, and results are interpreted within each task and revision.",
      "summary": "任务、环境与核验反馈",
      "summary_en": "Tasks, environments, and verification feedback",
      "url": "/lines/rl-sft/",
      "accent": "green",
      "highlights": [
        "终端与软件工程",
        "客户支持与专业办公",
        "桌面与浏览器操作",
        "来源环境材料",
        "独立核验反馈"
      ],
      "highlights_en": [
        "Terminal and software engineering",
        "Customer support and professional work",
        "Desktop and browser interaction",
        "Source environment materials",
        "Independent verification feedback"
      ],
      "proof_metrics": [
        {
          "label": "公开环境 / 任务",
          "label_en": "Public environments / tasks",
          "value": "43",
          "hint": "当前样例口径",
          "hint_en": "Current public sample scope"
        },
        {
          "label": "Zod 实测",
          "label_en": "Zod measured rollout",
          "value": "118/207",
          "hint": "Opus · Gold 207/207",
          "hint_en": "Opus · Gold 207/207"
        },
        {
          "label": "Simulator",
          "value": "0.35 → 1.0",
          "hint": "starter · oracle"
        },
        {
          "label": "企业工作流",
          "label_en": "Enterprise workflows",
          "value": "47–75%",
          "hint": "上游组级 best",
          "hint_en": "Upstream group-best range"
        },
        {
          "label": "安全运营",
          "label_en": "Security operations",
          "value": "0.149→0.764",
          "hint": "同 seed 轨迹覆盖",
          "hint_en": "Same-seed trajectory coverage"
        }
      ],
      "dataset_ids": [
        "gr-env-terminalbench4-001",
        "gr-env-deepswe-001",
        "gr-env-automationbench-001",
        "gr-env-jobbench-001",
        "gr-env-computer-use-001",
        "gr-env-clawbench-001",
        "gr-env-clawbench-wps-001",
        "gr-env-cyber-rf-vulhub-001",
        "gr-env-cyber-001",
        "gr-env-gdpevo-001",
        "gr-env-simulator-001",
        "gr-env-qna-001"
      ],
      "dataset_count": 12,
      "download": {
        "filename": "greatruth-rl-sft-environment-all-samples.zip",
        "url": "/downloads/greatruth-rl-sft-environment-all-samples.zip",
        "media_type": "application/zip",
        "byte_size": 1324306783,
        "sha256": "582feae1e20eb2794e751c9df13680dd435c33c3b3594ac734d67b3e3d2202f4",
        "dataset_count": 12,
        "file_count": 49
      }
    },
    {
      "id": "stem_data",
      "slug": "stem",
      "code": "PRODUCT 03",
      "title": "STEM 数据",
      "title_en": "STEM Data",
      "eyebrow": "KNOWLEDGE / RUBRIC",
      "description": "HLE / GPQA / FrontierMath 难度带的博士级数理评测与长文本研究题，交付题目、完整解答、Rubric，以及适用的模型作答和逐项评分证据。",
      "description_en": "PhD-level STEM evaluations and long-context research problems in the HLE / GPQA / FrontierMath difficulty band. Deliveries include problems, complete solutions, rubrics, and, where applicable, model responses with item-level grading evidence.",
      "meta_description_en": "GreaTruth STEM samples covering PhD-level mathematics and physics diagnostics plus long-context research problems",
      "page_description_en": "Public samples cover PhD-level STEM evaluations and long-context research problems. HLE, GPQA, and FrontierMath indicate difficulty bands only. Problems, reference solutions, rubrics, model responses, and grading evidence are presented according to each data contract.",
      "summary": "博士题评测与长文本研究题",
      "summary_en": "PhD evaluations and long-context research problems",
      "url": "/lines/stem/",
      "accent": "blue",
      "highlights": [
        "HLE 难度带",
        "GPQA 难度带",
        "FrontierMath 难度带",
        "参考解",
        "Rubric"
      ],
      "highlights_en": [
        "HLE difficulty band",
        "GPQA difficulty band",
        "FrontierMath difficulty band",
        "Reference solutions",
        "Rubrics"
      ],
      "proof_metrics": [
        {
          "label": "数学诊断",
          "label_en": "Mathematics diagnostic",
          "value": "1/10",
          "hint": "当前公开模型作答",
          "hint_en": "Current public model response"
        },
        {
          "label": "物理诊断",
          "label_en": "Physics diagnostic",
          "value": "3.5/10",
          "hint": "当前公开模型作答",
          "hint_en": "Current public model response"
        },
        {
          "label": "公开题项",
          "label_en": "Public evaluation items",
          "value": "5",
          "hint": "诊断 2 · 长文本研究题 3",
          "hint_en": "2 diagnostics · 3 research problems"
        },
        {
          "label": "MP02 Rubric",
          "label_en": "MP02 rubrics",
          "value": "20 步",
          "value_en": "20 steps",
          "hint": "3 个长文本研究题",
          "hint_en": "3 long-context research problems"
        }
      ],
      "dataset_ids": [
        "gr-eval-math-phd-001",
        "gr-stem-math-mp02-001",
        "gr-eval-physics-phd-001"
      ],
      "dataset_count": 3,
      "download": {
        "filename": "greatruth-stem-data-all-samples.zip",
        "url": "/downloads/greatruth-stem-data-all-samples.zip",
        "media_type": "application/zip",
        "byte_size": 133932,
        "sha256": "dd28d188e3abdc3554b269e329a60464fbca3bc688f8ed9ae784982734f62c0c",
        "dataset_count": 3,
        "file_count": 8
      }
    },
    {
      "id": "real_trajectory",
      "slug": "real-trajectory",
      "code": "PRODUCT 04",
      "title": "真实轨迹",
      "title_en": "Real Trajectories",
      "eyebrow": "BEHAVIOR / REALITY",
      "description": "Claude Code、OpenClaw 与 Hermes 的脱敏生产会话，按 SFT 轨迹交付任务上下文、CoT、工具调用、子代理过程、环境响应和最终输出。",
      "description_en": "De-identified production sessions from Claude Code, OpenClaw, and Hermes, delivered as SFT trajectories with task context, CoT, tool calls, subagent activity, environment responses, and final outputs.",
      "meta_description_en": "GreaTruth real-trajectory samples covering tool use and multi-agent execution in Claude Code, OpenClaw, and Hermes",
      "page_description_en": "De-identified real trajectories cover Claude Code, OpenClaw, and Hermes. Sessions are reviewed for task completeness, tool results, error recovery, subagent merges, and final state.",
      "summary": "真实会话轨迹",
      "summary_en": "Real session trajectories",
      "url": "/lines/real-trajectory/",
      "accent": "orange",
      "highlights": [
        "SFT 轨迹",
        "完整 CoT",
        "工具调用",
        "子代理协作",
        "生产协议"
      ],
      "highlights_en": [
        "SFT trajectories",
        "Complete CoT",
        "Tool calls",
        "Subagent collaboration",
        "Production protocols"
      ],
      "proof_metrics": [
        {
          "label": "公开轨迹",
          "label_en": "Public trajectories",
          "value": "4",
          "hint": "真实会话样例",
          "hint_en": "Real-session samples"
        },
        {
          "label": "规范化步骤",
          "label_en": "Normalized steps",
          "value": "478",
          "hint": "四条轨迹合计",
          "hint_en": "Total across four trajectories"
        },
        {
          "label": "工具调用",
          "label_en": "Tool calls",
          "value": "348",
          "hint": "来源协议统计",
          "hint_en": "Source-protocol count"
        },
        {
          "label": "子代理",
          "label_en": "Subagents",
          "value": "39",
          "hint": "OpenClaw / Hermes"
        },
        {
          "label": "最长用户轮",
          "label_en": "Longest user session",
          "value": "42",
          "hint": "Claude Code",
          "hint_en": "Claude Code user turns"
        }
      ],
      "dataset_ids": [
        "gr-agent-cc-001",
        "gr-agent-cc-002",
        "gr-agent-openclaw-001",
        "gr-agent-hermes-001"
      ],
      "dataset_count": 4,
      "download": {
        "filename": "greatruth-real-trajectory-all-samples.zip",
        "url": "/downloads/greatruth-real-trajectory-all-samples.zip",
        "media_type": "application/zip",
        "byte_size": 71674724,
        "sha256": "eb98afa58025373a6a9561e5590c9bbcebb09a8a02932aa5b4dc9516a84afdbb",
        "dataset_count": 4,
        "file_count": 17
      }
    },
    {
      "id": "high_quality_trajectory",
      "slug": "high-quality-trajectory",
      "code": "PRODUCT 05",
      "title": "高质量轨迹数据",
      "title_en": "High-quality Trajectory Data",
      "eyebrow": "SYNTHETIC / ENVIRONMENT",
      "description": "面向 Agent 后训练的合成 SFT 轨迹，覆盖 MCP 多工具环境任务与 WorkBuddy 多智能体任务树；保留完整 System Prompt、CoT、工具结果、运行状态和最终工作区。",
      "description_en": "Synthetic SFT trajectories for agent post-training, covering MCP multi-tool environment tasks and WorkBuddy multi-agent task trees. Complete system prompts, CoT, tool results, run states, and final workspaces are retained when captured by the source.",
      "meta_description_en": "GreaTruth high-quality trajectory data for MCP multi-tool environments and WorkBuddy multi-agent collaboration",
      "page_description_en": "Synthetic SFT trajectories cover MCP multi-tool environments and WorkBuddy multi-agent collaboration, including environment state, tool execution, task trees, run status, and final workspaces. Complete source content and missing-field states can be verified in the dedicated viewers.",
      "summary": "环境与框架增强轨迹",
      "summary_en": "Environment- and framework-enriched trajectories",
      "url": "/lines/high-quality-trajectory/",
      "accent": "violet",
      "highlights": [
        "SFT 轨迹",
        "MCP 工具过程",
        "WorkBuddy 框架",
        "完整 CoT / System Prompt",
        "done / partial"
      ],
      "highlights_en": [
        "SFT trajectories",
        "MCP tool execution",
        "WorkBuddy framework",
        "Complete CoT / system prompts",
        "done / partial"
      ],
      "proof_metrics": [
        {
          "label": "公开轨迹",
          "label_en": "Public trajectories",
          "value": "29",
          "hint": "MCP 17 · WorkBuddy 12",
          "hint_en": "MCP 17 · WorkBuddy 12"
        },
        {
          "label": "MCP 任务",
          "label_en": "MCP tasks",
          "value": "17/17",
          "hint": "307/307 checks"
        },
        {
          "label": "WorkBuddy 状态",
          "label_en": "WorkBuddy status",
          "value": "8 / 4",
          "hint": "done / partial"
        },
        {
          "label": "System Prompt",
          "value": "10/12",
          "hint": "完整捕获",
          "hint_en": "Captured in full"
        },
        {
          "label": "协作要求",
          "label_en": "Collaboration requirement",
          "value": "7/8",
          "hint": "达到最低要求",
          "hint_en": "Minimum requirement met"
        }
      ],
      "dataset_ids": [
        "gr-synthetic-mcp-001",
        "gr-synthetic-mixed-001"
      ],
      "dataset_count": 2,
      "download": {
        "filename": "greatruth-high-quality-trajectory-all-samples.zip",
        "url": "/downloads/greatruth-high-quality-trajectory-all-samples.zip",
        "media_type": "application/zip",
        "byte_size": 132603805,
        "sha256": "e0e35a3b4f765ba927805898110a12aaea8c2b2ce750da8c2f820dac2247f4cc",
        "dataset_count": 2,
        "file_count": 4
      }
    }
  ],
  "endpoints": {
    "catalog": "/data/catalog.json",
    "agent_instructions": "/llms.txt",
    "agent_manifest": "/ai-data.json",
    "trajectory_schema": "/data/schemas/trajectory.schema.json",
    "claude_api_turn_schema": "/data/schemas/claude-api-turn.schema.json",
    "openclaw_event_schema": "/data/schemas/openclaw-event.schema.json",
    "hermes_schema": "/data/schemas/hermes-trajectory.schema.json",
    "hermes_skill_evolution_schema": "/data/schemas/hermes-skill-evolution.schema.json",
    "evaluation_schema": "/data/schemas/qa-eval.schema.json",
    "verified_qa_schema": "/data/schemas/qa-verified.schema.json",
    "selection_schema": "/data/schemas/sample-selection.schema.json",
    "harbor_schema": "/data/schemas/harbor-env.schema.json",
    "harbor_samples": "/data/harbor/package-meta.json",
    "mcp_synthetic_schema": "/data/schemas/mcp-synthetic-traces.schema.json",
    "mixed_collaborative_schema": "/data/schemas/mixed-collaborative-trajectories.schema.json"
  },
  "summary": {
    "dataset_count": 30,
    "product_line_count": 5,
    "trace_step_count": 478,
    "evaluation_item_count": 5,
    "environment_item_count": 67,
    "sft_trajectory_count": 33,
    "synthetic_trajectory_count": 29,
    "downloadable_sample_count": 30
  },
  "datasets": [
    {
      "id": "gr-agent-cc-001",
      "legacy_id": "cc",
      "slug": "claude-code-trajectory",
      "title": "Claude Code 轨迹",
      "title_en": "Claude Code Agent Trajectory",
      "eyebrow": "CLAUDE CODE · API TURN",
      "description": "一次真实 coding 会话：103 次 API 轮次、83 次工具调用，保留任务上下文与工具结果，便于复核执行过程。",
      "description_en": "A real coding-agent session with 103 API turns and 83 tool calls. Task context and tool results are retained for end-to-end execution review.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "生产 Coding Agent 协议",
        "label_en": "Production coding-agent protocol",
        "scope": "真实 coding 会话",
        "scope_en": "Real coding session"
      },
      "category": "real_trajectory",
      "visual_kind": "linear_trace",
      "version": "2026.07",
      "release_date": "2026-07-15",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "reasoning",
        "tool_call",
        "tool_result",
        "output"
      ],
      "tags": [
        "Claude Code",
        "Coding",
        "Tool Use"
      ],
      "viewer_url": "/claude-code/viewer.html",
      "schema_url": "/data/schemas/claude-api-turn.schema.json",
      "quality": {
        "metric": "api_turns",
        "label": "API 轮次",
        "label_en": "API turns",
        "display": "103",
        "value": 103,
        "sort_value": 83,
        "status_label": "过程深度",
        "status_label_en": "Process depth",
        "basis": "lowest tool error rate in the source collection"
      },
      "preview": {
        "sequence": [
          "user",
          "tool",
          "tool",
          "tool",
          "output"
        ],
        "metrics": [
          {
            "label": "API 轮次",
            "label_en": "API turns",
            "value": "103"
          },
          {
            "label": "工具调用",
            "label_en": "Tool calls",
            "value": "83"
          },
          {
            "label": "Token 估算",
            "label_en": "Estimated tokens",
            "value": "123.8K"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "step schema",
          "tool call-result pairing",
          "execution completeness",
          "final output completeness"
        ]
      },
      "download": {
        "filename": "greatruth-claude-code-high-quality.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-claude-code-high-quality.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-high-quality.zip",
        "byte_size": 6416632,
        "sha256": "dd1605c53ee4608a57663eb13a7b2c140c638137948651988c66d87855c0790d",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "real_trajectory",
      "content": {
        "inventory": [
          "103 次 API 轮次，83 次工具调用",
          "含推理过程、工具入参与返回、最终输出",
          "配套规范化轨迹与下载样例"
        ],
        "inventory_en": [
          "103 API turns and 83 tool calls",
          "Reasoning, tool inputs and results, and the final output",
          "Normalized trajectory and downloadable source sample"
        ]
      },
      "record_count": 103,
      "record_unit": "api_turns",
      "sample_count": 1,
      "step_count": 124,
      "user_query": "先看下 src/api.mjs 里 set 函数的 for...in 循环，`if (!attributes[attributeName])` 那块会直接跳过值为 0、false 或空字符串的属性。再去 test/tests.js 里查一下有没有覆盖这些情况的用例。",
      "observability": {
        "source_event_count": 103,
        "normalized_event_count": 124,
        "tool_call_count": 83,
        "subagent_call_count": 0,
        "token_usage": {
          "available": true,
          "estimated": true,
          "source_usage_available": true,
          "token_unit": "estimated_tokens",
          "input": 2,
          "output": 164,
          "cache_read": 123252,
          "cache_write": 408,
          "total": 123826,
          "usage_records": 103,
          "source": "response.response_data.usage",
          "total_basis": "last available usage record",
          "main_trajectory": 123826,
          "subtrajectories": 0,
          "normalized_content_estimate": {
            "main_trajectory": 44576,
            "subtrajectories": 0,
            "total": 44576
          },
          "estimation_method": "last-turn usage total"
        }
      },
      "selection": {
        "mode": "lowest_tool_error_rate",
        "candidate_count": 10,
        "selected_id": "62732ab4-0bc4-447a-8e1c-c99e58abbbd6",
        "report_url": "/data/selections/claude-code-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Selected raw trajectory sample",
          "filename": "greatruth-claude-code-high-quality.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/claude-api-turn.schema.json",
          "url": "/downloads/greatruth-claude-code-high-quality.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-high-quality.jsonl",
          "byte_size": 15182517,
          "sha256": "34f0594afd71391ddc48cf5184f3d826bba118b673c9d0a39d343dd822219d0a",
          "record_count": 103,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "normalized_trace",
          "title": "Normalized Claude Code API-turn trace",
          "filename": "greatruth-claude-code-normalized.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/trajectory.schema.json",
          "url": "/downloads/greatruth-claude-code-normalized.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-normalized.json",
          "byte_size": 266579,
          "sha256": "88f5363a7a723284d3a55de90ddebd5dd6ec87923380fd2ebf0fd907b685163d",
          "record_count": 124,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Sample selection report",
          "filename": "claude-code-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/claude-code-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/claude-code-selection.json",
          "byte_size": 5532,
          "sha256": "dc6af2bd26ca81c83bbb4a917a7daf945e1b8be8a57ddcfe3b0d061f1d0f1b36",
          "record_count": 10,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-agent-cc-002",
      "legacy_id": "cc-40turn",
      "slug": "claude-code-40turn-trajectory",
      "title": "Claude Code 40 轮轨迹",
      "title_en": "Claude Code 40-turn Agent Trajectory",
      "eyebrow": "CLAUDE CODE · 40-TURN",
      "description": "一次长程 coding 会话：42 轮用户对话、131 次 API 轮次、108 次工具调用。",
      "description_en": "A long-horizon coding-agent session with 42 user turns, 131 API turns, and 108 tool calls.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "生产 Coding Agent 协议",
        "label_en": "Production coding-agent protocol",
        "scope": "长程 coding 会话",
        "scope_en": "Long-horizon coding session"
      },
      "category": "real_trajectory",
      "visual_kind": "linear_trace",
      "version": "2026.07.20",
      "release_date": "2026-07-20",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "reasoning",
        "tool_call",
        "tool_result",
        "output"
      ],
      "tags": [
        "Claude Code",
        "Coding",
        "Tool Use",
        "40-turn"
      ],
      "viewer_url": "/claude-code-40turn/viewer.html",
      "schema_url": "/data/schemas/claude-api-turn.schema.json",
      "quality": {
        "metric": "user_turns",
        "label": "用户轮次",
        "label_en": "User turns",
        "display": "42",
        "value": 42,
        "sort_value": 108,
        "status_label": "过程深度",
        "status_label_en": "Process depth",
        "basis": "lowest tool error rate in the source collection"
      },
      "preview": {
        "sequence": [
          "user",
          "tool",
          "tool",
          "tool",
          "output"
        ],
        "metrics": [
          {
            "label": "用户轮次",
            "label_en": "User turns",
            "value": "42"
          },
          {
            "label": "API 轮次",
            "label_en": "API turns",
            "value": "131"
          },
          {
            "label": "工具调用",
            "label_en": "Tool calls",
            "value": "108"
          },
          {
            "label": "Token 估算",
            "label_en": "Estimated tokens",
            "value": "218.2K"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "step schema",
          "tool call-result pairing",
          "execution completeness",
          "final output completeness"
        ]
      },
      "download": {
        "filename": "greatruth-claude-code-40turn-high-quality.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-claude-code-40turn-high-quality.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-40turn-high-quality.zip",
        "byte_size": 384072,
        "sha256": "1d7f23f4e91fcc8a14618fdbf7f2093f418fa8d593cf55bb42521d5bc6538a51",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "real_trajectory",
      "content": {
        "inventory": [
          "42 轮用户对话，108 次工具调用",
          "长程会话，工具往返过程完整保留",
          "配套规范化轨迹与下载样例"
        ],
        "inventory_en": [
          "42 user turns and 108 tool calls",
          "Long-horizon session with complete tool exchanges",
          "Normalized trajectory and downloadable source sample"
        ]
      },
      "record_count": 131,
      "record_unit": "api_turns",
      "sample_count": 1,
      "step_count": 174,
      "user_query": "先用 nox 跑一下现有测试，针对 Python 3.13 的 tests session，确认全部通过。如果跑不起来就告诉我。",
      "observability": {
        "source_event_count": 131,
        "normalized_event_count": 174,
        "tool_call_count": 108,
        "subagent_call_count": 0,
        "token_usage": {
          "available": true,
          "estimated": true,
          "source_usage_available": true,
          "token_unit": "estimated_tokens",
          "input": 2,
          "output": 1950,
          "cache_read": 214583,
          "cache_write": 1671,
          "total": 218206,
          "usage_records": 131,
          "source": "response.response_data.usage",
          "total_basis": "last available usage record",
          "main_trajectory": 218206,
          "subtrajectories": 0,
          "normalized_content_estimate": {
            "main_trajectory": 93646,
            "subtrajectories": 0,
            "total": 93646
          },
          "estimation_method": "last-turn usage total"
        }
      },
      "selection": {
        "mode": "lowest_tool_error_rate",
        "candidate_count": 10,
        "selected_id": "64e58ff6-14fa-44bd-a0ff-02bce1e07c11",
        "report_url": "/data/selections/claude-code-40turn-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Selected raw trajectory sample",
          "filename": "greatruth-claude-code-40turn-high-quality.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/claude-api-turn.schema.json",
          "url": "/downloads/greatruth-claude-code-40turn-high-quality.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-40turn-high-quality.jsonl",
          "byte_size": 28734528,
          "sha256": "f1c51b44d868963081f796827ea51a34efd4ed603ff8a511c53edecb37c6a4e1",
          "record_count": 131,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "normalized_trace",
          "title": "Normalized Claude Code API-turn trace",
          "filename": "greatruth-claude-code-40turn-normalized.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/trajectory.schema.json",
          "url": "/downloads/greatruth-claude-code-40turn-normalized.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-claude-code-40turn-normalized.json",
          "byte_size": 466616,
          "sha256": "48f9b4e63f110522a2f3b0bd5fe1c769b3bc3928c9a200b3415c76788516ba6c",
          "record_count": 174,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Sample selection report",
          "filename": "claude-code-40turn-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/claude-code-40turn-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/claude-code-40turn-selection.json",
          "byte_size": 5867,
          "sha256": "17bd51994661d8f85cd267081aadc3e476310b3a44c64d91aad60d710c2fc6c0",
          "record_count": 10,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-agent-openclaw-001",
      "legacy_id": "openclaw-000125",
      "slug": "openclaw-multi-agent-trajectory",
      "title": "OpenClaw 多代理轨迹",
      "title_en": "OpenClaw Multi-agent Trajectory",
      "eyebrow": "OPENCLAW · MULTI-AGENT",
      "description": "一次多代理协作轨迹，保留派发、回收与合并过程。",
      "description_en": "A multi-agent collaboration trajectory retaining delegation, result collection, and merge operations.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "生产 Multi-agent 协议",
        "label_en": "Production multi-agent protocol",
        "scope": "多代理协作会话",
        "scope_en": "Multi-agent collaboration session"
      },
      "category": "real_trajectory",
      "visual_kind": "branched_trace",
      "version": "2026.07",
      "release_date": "2026-07-15",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "reasoning",
        "tool_call",
        "subagent",
        "merge",
        "output"
      ],
      "tags": [
        "OpenClaw",
        "Multi-agent",
        "Long Horizon"
      ],
      "viewer_url": "/openclaw/viewer.html",
      "schema_url": "/data/schemas/openclaw-event.schema.json",
      "quality": {
        "metric": "subagent_calls",
        "label": "子代理调用",
        "label_en": "Subagent calls",
        "display": "8",
        "value": 8,
        "sort_value": 52,
        "status_label": "过程深度",
        "status_label_en": "Process depth",
        "basis": "highest structural completeness in the source collection"
      },
      "preview": {
        "sequence": [
          "user",
          "tool",
          "subagent",
          "subagent",
          "merge",
          "output"
        ],
        "metrics": [
          {
            "label": "子代理派发",
            "label_en": "Subagent dispatches",
            "value": "3"
          },
          {
            "label": "结果回收",
            "label_en": "Merged results",
            "value": "6"
          },
          {
            "label": "原始事件",
            "label_en": "Source events",
            "value": "128"
          },
          {
            "label": "工具调用",
            "label_en": "Tool calls",
            "value": "52"
          },
          {
            "label": "子代理",
            "label_en": "Subagents",
            "value": "8"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "step schema",
          "subagent fan-out",
          "merge recovery",
          "tool call-result pairing"
        ]
      },
      "download": {
        "filename": "greatruth-openclaw-high-quality.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-openclaw-high-quality.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-openclaw-high-quality.zip",
        "byte_size": 2304128,
        "sha256": "705be286e93c9e6255bec419c24be2a6995d693df71fee593c274d626ffa64a8",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "real_trajectory",
      "content": {
        "inventory": [
          "8 次子代理调用，52 次工具调用",
          "含派发、回收与结果合并过程",
          "配套规范化轨迹与下载样例"
        ],
        "inventory_en": [
          "8 subagent calls and 52 tool calls",
          "Delegation, result collection, and merge operations",
          "Normalized trajectory and downloadable source sample"
        ]
      },
      "record_count": 128,
      "record_unit": "events",
      "sample_count": 1,
      "step_count": 60,
      "user_query": "我刚收到上游团队的反馈草案，指南主稿、快速开始和审核清单边界确实没拉开，快速开始里连最低 Go 版本都没写。你先完整读一遍 input_materials/origin/ 下的所有材料（material_context.json、material_manifest.json 以及列出的 origin 文件），再基于材料里的 known_gaps 和 key_entities 做以下事：用表格加注释的方式，清晰界定三部分各自覆盖什么、重叠什么、缺口在哪；针对 quickstart 缺的 Go 版本要求（Go 1.22+？），如果材料里没明确写，可以用 bocha_web_search 快速搜一下当前主流团队建议的最低版本并记录来源 URL。先把分析写到 deliverables/boundary_analysis.md，要求包含 key_entities（go test -count=1 ./ 等命令要不要补到 quickstart 里）和每个缺口的处置建议。先在内部走一遍推理、对照材料约束再落盘，输出可复查。",
      "observability": {
        "source_event_count": 128,
        "normalized_event_count": 60,
        "tool_call_count": 52,
        "subagent_call_count": 8,
        "token_usage": {
          "available": true,
          "estimated": true,
          "source_usage_available": true,
          "token_unit": "estimated_tokens",
          "input": 66694,
          "output": 320,
          "cache_read": 19456,
          "cache_write": 0,
          "total": 86470,
          "usage_records": 45,
          "source": "message.usage",
          "total_basis": "last available usage record",
          "main_trajectory": 74037,
          "subtrajectories": 12433,
          "normalized_content_estimate": {
            "main_trajectory": 79827,
            "subtrajectories": 13405,
            "total": 93232
          },
          "estimation_method": "last-turn usage total partitioned by normalized main/subtrajectory content length"
        }
      },
      "selection": {
        "mode": "highest_data_quality",
        "candidate_count": 5,
        "selected_id": "000125",
        "report_url": "/data/selections/openclaw-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Selected raw trajectory sample",
          "filename": "greatruth-openclaw-high-quality.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/openclaw-event.schema.json",
          "url": "/downloads/greatruth-openclaw-high-quality.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-openclaw-high-quality.jsonl",
          "byte_size": 545034,
          "sha256": "a5d5b7f42586af43a200f9e8e3e9eb6b1218066b41ecfa53a41c240eac6e97a4",
          "record_count": 128,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "normalized_trace",
          "title": "Normalized OpenClaw multi-agent event trace",
          "filename": "greatruth-openclaw-normalized.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/trajectory.schema.json",
          "url": "/downloads/greatruth-openclaw-normalized.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-openclaw-normalized.json",
          "byte_size": 424714,
          "sha256": "612a2162e835d7d3478637593188913c77d6e550ffc4be913aef08bfc4761eea",
          "record_count": 60,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Sample selection report",
          "filename": "openclaw-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/openclaw-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/openclaw-selection.json",
          "byte_size": 1918,
          "sha256": "85fb7c1dfb53347e679cfacf49d92b16e781687ad79d2af96d02417ae1638d32",
          "record_count": 5,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-agent-hermes-001",
      "legacy_id": "hermes-00002",
      "slug": "hermes-function-calling-trajectory",
      "title": "Hermes Function Calling 轨迹",
      "title_en": "Hermes Function Calling Trajectories",
      "eyebrow": "HERMES · FUNCTION CALLING",
      "description": "一次函数调用训练轨迹，含子任务委托与 12 条 skill evolution 事件。",
      "description_en": "A function-calling training trajectory with subtask delegation and 12 Skill Evolution events.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "Function Calling / Tool-use",
        "label_en": "Function calling / tool use",
        "scope": "训练数据形态",
        "scope_en": "Training-data format"
      },
      "category": "real_trajectory",
      "visual_kind": "branched_trace",
      "version": "2026.07",
      "release_date": "2026-07-15",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "reasoning",
        "tool_call",
        "tool_response",
        "subagent",
        "merge",
        "skill_evolution"
      ],
      "tags": [
        "Hermes",
        "Function Calling",
        "Skill Evolution"
      ],
      "viewer_url": "/hermes/viewer.html",
      "schema_url": "/data/schemas/hermes-trajectory.schema.json",
      "quality": {
        "metric": "skill_evolution",
        "label": "Skill 演化",
        "label_en": "Skill evolution",
        "display": "12",
        "value": 12,
        "sort_value": 105,
        "status_label": "过程深度",
        "status_label_en": "Process depth",
        "basis": "highest structural completeness in the source collection"
      },
      "preview": {
        "sequence": [
          "user",
          "tool",
          "subagent",
          "subagent",
          "merge",
          "output"
        ],
        "metrics": [
          {
            "label": "去重调用",
            "label_en": "Deduplicated calls",
            "value": "105"
          },
          {
            "label": "Skill 变更",
            "label_en": "Skill changes",
            "value": "12"
          },
          {
            "label": "对话事件",
            "label_en": "Conversation events",
            "value": "193"
          },
          {
            "label": "工具调用",
            "label_en": "Tool calls",
            "value": "105"
          },
          {
            "label": "子代理",
            "label_en": "Subagents",
            "value": "31"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "cumulative sample order",
          "tool call-response pairing",
          "delegate task fan-out",
          "state transition integrity",
          "skill evolution traceability"
        ]
      },
      "download": {
        "filename": "greatruth-hermes-trajectory-samples.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-hermes-trajectory-samples.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-hermes-trajectory-samples.zip",
        "byte_size": 9743780,
        "sha256": "73b8c88d695b8b9cf828aa23c1b5ac5a5d73d251698dddb77f3f4198794f0579",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "real_trajectory",
      "content": {
        "inventory": [
          "105 次工具调用，12 条 Skill 演化记录",
          "含子任务委托与结果合并",
          "配套规范化轨迹与下载样例"
        ],
        "inventory_en": [
          "105 tool calls and 12 Skill Evolution records",
          "Subtask delegation and result merges",
          "Normalized trajectory and downloadable source sample"
        ]
      },
      "record_count": 11,
      "record_unit": "trajectories",
      "sample_count": 1,
      "step_count": 120,
      "user_query": "For same-day handoff, I need you to close the remaining source-validation risk rather than adding more caveat files: in `/data/tasks/00002/output`, make one conservative attempt to replace or clearly quarantine the illustrative series by using saved public evidence from a mix of official, industry, and open-data domains, prioritizing the official FRED/ALFRED path via bocha-search and saving raw evidence; if a lightweight helper such as a small Python package or CLI fetch/parsing tool is useful, bootstrap it in the workspace and verify it with a minimal command. Please produce a compact table-first source-validation report with evidence notes, a quantitative checkpoint comparing any retrieved official/open-data observations against the prepared CSV, and a short narrative recommendation on whether the existing charts can stand as internal prototype only or should be regenerated; keep all outputs under the same directory, update only the files needed to make the conclusion traceable, and read back the key report plus any changed chart/data/manifest files before you claim it is ready.",
      "observability": {
        "source_event_count": 193,
        "normalized_event_count": 120,
        "tool_call_count": 105,
        "subagent_call_count": 31,
        "token_usage": {
          "available": true,
          "estimated": true,
          "source_usage_available": false,
          "token_unit": "estimated_tokens",
          "input": null,
          "output": null,
          "cache_read": null,
          "cache_write": null,
          "total": 202946,
          "main_trajectory": 131886,
          "subtrajectories": 71060,
          "usage_records": 0,
          "source": "normalized_final_trajectory_content",
          "total_basis": "final cumulative normalized trajectory",
          "normalized_content_estimate": {
            "main_trajectory": 131886,
            "subtrajectories": 71060,
            "total": 202946
          },
          "estimation_method": "CJK characters + non-CJK characters / 4"
        }
      },
      "selection": {
        "mode": "highest_data_quality",
        "candidate_count": 3,
        "selected_id": "00002",
        "report_url": "/data/selections/hermes-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Selected raw trajectory sample",
          "filename": "greatruth-hermes-trajectory-samples.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/hermes-trajectory.schema.json",
          "url": "/downloads/greatruth-hermes-trajectory-samples.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-hermes-trajectory-samples.jsonl",
          "byte_size": 6117423,
          "sha256": "4eeed4c7b122125cad739a80bafcd3a52db9ff849a06dc990ad7da7c5130f841",
          "record_count": 11,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "normalized_trace",
          "title": "Normalized Hermes function-calling trace",
          "filename": "greatruth-hermes-trajectory-normalized.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/trajectory.schema.json",
          "url": "/downloads/greatruth-hermes-trajectory-normalized.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-hermes-trajectory-normalized.json",
          "byte_size": 1024803,
          "sha256": "08d0c4f089fa6d794300f7c3118051bfadeea04da2a0ab154f0a94b54ade1e15",
          "record_count": 120,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Sample selection report",
          "filename": "hermes-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/hermes-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/hermes-selection.json",
          "byte_size": 1358,
          "sha256": "d7f4bdedb35fc761bba7908db69261ee8ae3671c5d2d133976972ba77923bfdb",
          "record_count": 3,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "skill_evolution",
          "title": "Hermes skill evolution events",
          "filename": "greatruth-hermes-skill-evolution-samples.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/hermes-skill-evolution.schema.json",
          "url": "/downloads/greatruth-hermes-skill-evolution-samples.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-hermes-skill-evolution-samples.jsonl",
          "byte_size": 45004,
          "sha256": "8f519dea487f36bffabca124d02b463b89ee1e968bdf5f183b6358dec49309e2",
          "record_count": 12,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-eval-math-phd-001",
      "legacy_id": "math-phd",
      "slug": "math-phd-evaluation",
      "title": "数学博士题评测",
      "title_en": "PhD Mathematics Evaluation",
      "eyebrow": "MATH · PhD",
      "description": "数学博士题一则。模型得分 1/10；样例保留题干、参考解、分步 Rubric 与评分说明。",
      "description_en": "One PhD-level mathematics problem with a model score of 1/10. The sample includes the full problem, reference solution, stepwise rubric, and grading rationale.",
      "buyer_type": "评测集 + Rubric",
      "buyer_type_en": "Evaluation set + rubric",
      "benchmark_anchor": {
        "label": "HLE / GPQA / FrontierMath 难度带",
        "label_en": "HLE / GPQA / FrontierMath difficulty band",
        "scope": "难度对标",
        "scope_en": "Difficulty reference"
      },
      "category": "stem_data",
      "visual_kind": "eval_matrix",
      "version": "2026.07",
      "release_date": "2026-07-15",
      "status": "verified",
      "languages": [
        "zh-CN"
      ],
      "modalities": [
        "problem",
        "reference_solution",
        "rubric",
        "model_answer",
        "grading"
      ],
      "tags": [
        "Math PhD",
        "Rubric",
        "Reasoning"
      ],
      "viewer_url": "/math-phd/viewer.html",
      "schema_url": "/data/schemas/qa-eval.schema.json",
      "quality": {
        "metric": "selected_ai_score",
        "label": "AI 作答",
        "label_en": "Model response",
        "score": 1,
        "scale": 10,
        "basis": "global minimum AI response score in the source collection"
      },
      "preview": {
        "values": [
          1
        ],
        "metrics": [
          {
            "label": "AI 作答得分",
            "label_en": "AI response score",
            "value": "1/10"
          },
          {
            "label": "候选作答",
            "label_en": "Candidate responses",
            "value": "9"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "problem completeness",
          "reference-grounded rubric",
          "grading traceability",
          "formula fidelity"
        ]
      },
      "download": {
        "filename": "greatruth-math-low-score-evaluation.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-math-low-score-evaluation.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-math-low-score-evaluation.zip",
        "byte_size": 22974,
        "sha256": "6da6c7630ea0c073f7e08b121115c531cdbe733f886164f6f6bea5885ecb954f",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "stem_data",
      "content": {
        "label": "题干",
        "label_en": "Problem",
        "excerpt": "Dedekind-finite 环上的 pseudo-quadratic module：给定轨道描述，构造显式元素 g∉U₅U₁U₅H₁。要求构造性证明，不能用纯存在性代替。",
        "excerpt_en": "Pseudo-quadratic modules over a Dedekind-finite ring: construct an explicit element g∉U₅U₁U₅H₁ from the orbit description. A constructive proof is required; a purely existential argument is insufficient.",
        "inventory": [
          "题干、参考解、分步 Rubric 与评分说明齐全",
          "保留模型作答与逐项评语",
          "用于复核推理缺口的诊断样例"
        ],
        "inventory_en": [
          "Complete problem, reference solution, stepwise rubric, and grading rationale",
          "Model response and item-level grader comments",
          "Diagnostic sample for reviewing reasoning gaps"
        ]
      },
      "record_count": 1,
      "record_unit": "items",
      "sample_count": 1,
      "selection": {
        "mode": "lowest_ai_score",
        "candidate_count": 9,
        "selected_id": "mp01-003-doubao-1",
        "report_url": "/data/selections/math-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Evaluation sample JSON",
          "filename": "greatruth-math-low-score-evaluation.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/qa-eval.schema.json",
          "url": "/downloads/greatruth-math-low-score-evaluation.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-math-low-score-evaluation.json",
          "byte_size": 9327,
          "sha256": "053af847c3420f090476a79deb729a0c8b4c472af0f464132bd5a6d09e9d8d8d",
          "record_count": 1,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Evaluation sample selection report",
          "filename": "math-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/math-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/math-selection.json",
          "byte_size": 1615,
          "sha256": "48e8cfb751c0af6224202fdc79434ecf48be02a1f6ff9bd7ec2ca8acfaa5a42f",
          "record_count": 9,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-stem-math-mp02-001",
      "legacy_id": "math-mp02",
      "slug": "math-mp02-verified-questions",
      "title": "长文本研究题",
      "title_en": "Long-context Research Problems",
      "eyebrow": "MATH · MP02 VERIFIED",
      "description": "长上下文、科研课题级别研究问题，提供完整解答；包含完整题目、参考解、分步 Rubric 与知识标签。",
      "description_en": "Long-context, research-grade problems with complete solutions, stepwise rubrics, and subject tags.",
      "buyer_type": "评测集 + Rubric",
      "buyer_type_en": "Evaluation set + rubric",
      "benchmark_anchor": {
        "label": "HLE / GPQA / FrontierMath 难度带",
        "label_en": "HLE / GPQA / FrontierMath difficulty band",
        "scope": "难度对标",
        "scope_en": "Difficulty reference"
      },
      "category": "stem_data",
      "visual_kind": "eval_matrix",
      "version": "2026.07",
      "release_date": "2026-07-29",
      "status": "verified",
      "languages": [
        "zh-CN"
      ],
      "modalities": [
        "problem",
        "reference_solution",
        "rubric",
        "keywords"
      ],
      "tags": [
        "Long Context",
        "Research Problems",
        "Rubric"
      ],
      "viewer_url": "/math-mp02/viewer.html",
      "schema_url": "/data/schemas/qa-verified.schema.json",
      "quality": {
        "metric": "verified_records",
        "label": "全部核验",
        "label_en": "Fully verified",
        "display": "3 题",
        "display_en": "3 problems",
        "status_label": "题集样例",
        "status_label_en": "Problem-set sample",
        "basis": "all records passed source verification and contain the required fields"
      },
      "preview": {
        "metrics": [
          {
            "label": "题目",
            "label_en": "Problems",
            "value": "3"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric",
            "value": "20 步",
            "value_en": "20 steps"
          },
          {
            "label": "主题",
            "label_en": "Topics",
            "value": "3 类",
            "value_en": "3 categories"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "problem completeness",
          "solution consistency",
          "rubric coverage",
          "formula fidelity"
        ]
      },
      "download": {
        "filename": "greatruth-math-mp02-verified.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-math-mp02-verified.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-math-mp02-verified.zip",
        "byte_size": 10022,
        "sha256": "b725494fdb557592a8bea695b34b9e4708e0bce89bdbae4863f36e998e40869f",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "stem_data",
      "content": {
        "label": "题集范围",
        "label_en": "Problem-set scope",
        "excerpt": "长上下文科研课题级研究问题，覆盖有向拟阵、数论与算术几何、随机几何中的证明和计算任务，并提供完整解答。",
        "excerpt_en": "Long-context, research-grade proof and computation problems spanning oriented matroids, number theory and arithmetic geometry, and stochastic geometry, each with a complete solution.",
        "inventory": [
          "3 道长文本研究题",
          "题目、参考解、Rubric 与知识标签齐全",
          "共 20 个分步评分节点"
        ],
        "inventory_en": [
          "3 long-context research problems",
          "Complete problems, reference solutions, rubrics, and subject tags",
          "20 stepwise scoring criteria"
        ]
      },
      "record_count": 3,
      "record_unit": "items",
      "sample_count": 3,
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Verified mathematics question records",
          "filename": "greatruth-math-mp02-verified.jsonl",
          "media_type": "application/x-ndjson",
          "schema_url": "/data/schemas/qa-verified.schema.json",
          "url": "/downloads/greatruth-math-mp02-verified.jsonl",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-math-mp02-verified.jsonl",
          "byte_size": 32836,
          "sha256": "526f0ce5feb464cdb310ae8bc4c724c8ebb25883db7cd98fb792373febace095",
          "record_count": 3,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-eval-physics-phd-001",
      "legacy_id": "physics-phd",
      "slug": "physics-phd-evaluation",
      "title": "物理博士题评测",
      "title_en": "PhD Physics Evaluation",
      "eyebrow": "PHYSICS · PhD",
      "description": "物理博士题一则。模型得分 3.5/10；样例保留公式题干、参考解、分步 Rubric 与评分说明。",
      "description_en": "One PhD-level physics problem with a model score of 3.5/10. The sample includes the formula-rich problem, reference solution, stepwise rubric, and grading rationale.",
      "buyer_type": "评测集 + Rubric",
      "buyer_type_en": "Evaluation set + rubric",
      "benchmark_anchor": {
        "label": "HLE / GPQA 难度带",
        "label_en": "HLE / GPQA difficulty band",
        "scope": "难度对标",
        "scope_en": "Difficulty reference"
      },
      "category": "stem_data",
      "visual_kind": "eval_matrix",
      "version": "2026.07",
      "release_date": "2026-07-15",
      "status": "verified",
      "languages": [
        "zh-CN"
      ],
      "modalities": [
        "problem",
        "reference_solution",
        "rubric",
        "model_answer",
        "grading"
      ],
      "tags": [
        "Physics PhD",
        "Rubric",
        "Scientific Reasoning"
      ],
      "viewer_url": "/physics-phd/viewer.html",
      "schema_url": "/data/schemas/qa-eval.schema.json",
      "quality": {
        "metric": "selected_ai_score",
        "label": "AI 作答",
        "label_en": "Model response",
        "score": 3.5,
        "scale": 10,
        "basis": "global minimum AI response score in the source collection"
      },
      "preview": {
        "values": [
          3.5
        ],
        "metrics": [
          {
            "label": "AI 作答得分",
            "label_en": "AI response score",
            "value": "3.5/10"
          },
          {
            "label": "候选作答",
            "label_en": "Candidate responses",
            "value": "14"
          }
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "problem completeness",
          "reference-grounded rubric",
          "grading traceability",
          "formula fidelity"
        ]
      },
      "download": {
        "filename": "greatruth-physics-low-score-evaluation.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-physics-low-score-evaluation.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-physics-low-score-evaluation.zip",
        "byte_size": 37259,
        "sha256": "43a2c4cfe86bfb28813ea1b0da6023043daf8293bc90a554ae4e02d35c54e3de",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "stem_data",
      "content": {
        "label": "题干",
        "label_en": "Problem",
        "excerpt": "AdS–JT 双边黑洞中带观察者的 Hilbert 空间：论证微扰内积正定、非微扰正半定，以及把观察者当作 Gaussian 物质时空间坍缩为一维。",
        "excerpt_en": "Hilbert space with an observer in a two-sided AdS-JT black hole: establish perturbative positivity, non-perturbative positive semidefiniteness, and the collapse to one dimension when the observer is modeled as Gaussian matter.",
        "inventory": [
          "公式题干、标准答案、Rubric 与评分证据齐全",
          "保留模型作答与评分说明",
          "用于复核推理缺口的诊断样例"
        ],
        "inventory_en": [
          "Complete formula-rich problem, reference solution, rubric, and grading evidence",
          "Model response and grading rationale",
          "Diagnostic sample for reviewing reasoning gaps"
        ]
      },
      "record_count": 1,
      "record_unit": "items",
      "sample_count": 1,
      "selection": {
        "mode": "lowest_ai_score",
        "candidate_count": 14,
        "selected_id": "pp01-007-qwen3-5-plus",
        "report_url": "/data/selections/physics-selection.json"
      },
      "artifacts": [
        {
          "role": "raw_sample",
          "title": "Evaluation sample JSON",
          "filename": "greatruth-physics-low-score-evaluation.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/qa-eval.schema.json",
          "url": "/downloads/greatruth-physics-low-score-evaluation.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-physics-low-score-evaluation.json",
          "byte_size": 15771,
          "sha256": "c58e77ee295bcb83fd46a97ebc6dacd88c88d571b1d7ec51f3b38fd4a3b341cd",
          "record_count": 1,
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "selection_report",
          "title": "Evaluation sample selection report",
          "filename": "physics-selection.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/sample-selection.schema.json",
          "url": "/data/selections/physics-selection.json",
          "content_url": "https://display.greatruth.cloud/data/selections/physics-selection.json",
          "byte_size": 2074,
          "sha256": "4ba58d947633952cb3d6d1c33b82d6354f459d4796426b2d0a031a7273b521c7",
          "record_count": 14,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-env-terminalbench4-001",
      "slug": "terminalbench4",
      "title": "TerminalBench 4 · 终端任务",
      "title_en": "TerminalBench 4 · Terminal Tasks",
      "eyebrow": "TB4 · TERMINAL TASKS",
      "eyebrow_en": "TB4 · TERMINAL TASKS",
      "description": "终端任务样例覆盖发布门禁、指标聚合、订单履约、冷链审计、限流与优化器状态管理。每个来源包保留完整题面、实际提供的工作区与测试材料，以及运行轨迹和评测记录；不同任务版本与运行范围分别展示。",
      "description_en": "Terminal task samples cover release gates, metric aggregation, order fulfillment, cold-chain auditing, rate limiting, and optimizer state management. Each source package retains complete instructions, supplied workspace and test materials, run trajectories, and evaluation records, with task revisions and run scopes shown separately.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "TB4 来源集合",
        "label_en": "TB4 source collection",
        "scope": "上传目录中的终端任务包",
        "scope_en": "Terminal task packages in the uploaded collection"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.29",
      "release_date": "2026-09-29",
      "status": "sample",
      "languages": [
        "en",
        "zh"
      ],
      "modalities": [
        "instruction",
        "terminal",
        "workspace",
        "verifier",
        "trajectory"
      ],
      "tags": [
        "TerminalBench 4",
        "TB4",
        "Terminal",
        "RLVR"
      ],
      "viewer_url": "/terminalbench4/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按包核验",
        "display_en": "Per-package evidence",
        "basis": "Source revisions and run scopes remain separate; no site rerun",
        "status_label": "任务样例",
        "status_label_en": "Task sample"
      },
      "preview": {
        "label": "TERMINAL TASK",
        "label_en": "TERMINAL TASK",
        "summary": "终端执行与独立测试",
        "summary_en": "Terminal execution and independent tests",
        "sequence": [
          "instruction",
          "workspace",
          "execution",
          "verifier",
          "trajectory"
        ],
        "metrics": [
          {
            "label": "来源包",
            "label_en": "Source packages",
            "value": "10"
          }
        ],
        "note": "保留来源任务版本和评测分母，不合并为整体通过率。",
        "note_en": "Source revisions and evaluation denominators remain separate; no combined pass rate is inferred."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "终端任务样例覆盖发布门禁、指标聚合、订单履约、冷链审计、限流与优化器状态管理。每个来源包保留完整题面、实际提供的工作区与测试材料，以及运行轨迹和评测记录；不同任务版本与运行范围分别展示。",
        "excerpt_en": "Terminal task samples cover release gates, metric aggregation, order fulfillment, cold-chain auditing, rate limiting, and optimizer state management. Each source package retains complete instructions, supplied workspace and test materials, run trajectories, and evaluation records, with task revisions and run scopes shown separately.",
        "inventory": [
          "完整来源题面与终端工作区材料",
          "各包测试文件和来源评测摘要",
          "原始任务 ZIP 与运行轨迹"
        ],
        "inventory_en": [
          "Complete source instructions and terminal workspace materials",
          "Per-package tests and source evaluation summaries",
          "Original task ZIPs and run trajectories"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "source revision and evaluation denominator",
          "environment and verifier files",
          "archive integrity"
        ]
      },
      "record_count": 10,
      "record_unit": "packages",
      "sample_count": 10,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "TerminalBench 4 structured sample",
          "filename": "terminalbench4-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/terminalbench4-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/terminalbench4-sample.json",
          "byte_size": 520191,
          "sha256": "c450ade801fa500acab456eb2cee5a5ee7cfd231a2d1c7f154e8218fe1cbc36c",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "artifact-provenance-reconciler-v2",
          "filename": "greatruth-tb4-artifact-provenance-reconciler-v2.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-artifact-provenance-reconciler-v2.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-artifact-provenance-reconciler-v2.zip",
          "byte_size": 468096,
          "sha256": "f8c163f95d8f52a76ce150ee1b21745b2a54dff8a154583d0f487cd8ade7ed4d",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "cold-chain-custody-audit",
          "filename": "greatruth-tb4-cold-chain-custody-audit.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-cold-chain-custody-audit.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-cold-chain-custody-audit.zip",
          "byte_size": 465862,
          "sha256": "25113e28d47e30cfb02bc39a6c6b16638bd67e7d12b699dbf33820bd1086fc43",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "diagnostic-baseline-release-gate",
          "filename": "greatruth-tb4-diagnostic-baseline-release-gate.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-diagnostic-baseline-release-gate.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-diagnostic-baseline-release-gate.zip",
          "byte_size": 280061,
          "sha256": "a6b1ac5218fe0fad29eebececa8278b7b67bfdbc0d7801db4ad230f52f93b0b5",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "metric-aggregator",
          "filename": "greatruth-tb4-metric-aggregator.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-metric-aggregator.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-metric-aggregator.zip",
          "byte_size": 310756,
          "sha256": "424448501c64a9b919091106bc9d61cf482aee412d3a204eec9041bc3fb58d2d",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "order-fulfillment-engine",
          "filename": "greatruth-tb4-order-fulfillment-engine.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-order-fulfillment-engine.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-order-fulfillment-engine.zip",
          "byte_size": 172761,
          "sha256": "464a05abd06a5d9d615b714d462655822edefaf8702dd158c3fa5634e7a522a9",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "qc-astra-pass4-cmaes-r01-instruct3",
          "filename": "greatruth-tb4-qc-astra-pass4-cmaes-r01-instruct3.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-qc-astra-pass4-cmaes-r01-instruct3.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-qc-astra-pass4-cmaes-r01-instruct3.zip",
          "byte_size": 698673,
          "sha256": "dfd6d818a2b3efed1e3ffe876a4d3c2197f72d58ff38463d86b1e57911ed60da",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "qc-sol-pass4-cmaes-r01-instruct3-rev2",
          "filename": "greatruth-tb4-qc-sol-pass4-cmaes-r01-instruct3-rev2.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-qc-sol-pass4-cmaes-r01-instruct3-rev2.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-qc-sol-pass4-cmaes-r01-instruct3-rev2.zip",
          "byte_size": 1133873,
          "sha256": "ad8b91f4dc85434ec805a2bd408d8a6145ba8762a507dbf75bd3cba18a680c59",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "rate-limiter-gate",
          "filename": "greatruth-tb4-rate-limiter-gate.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-rate-limiter-gate.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-rate-limiter-gate.zip",
          "byte_size": 99399,
          "sha256": "9bf1c8e90080b7dd4a23fd7a19dafd0c83ca11fc79e8234a36ba196aa8c428c0",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "route-convergence-release-gate",
          "filename": "greatruth-tb4-route-convergence-release-gate.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-route-convergence-release-gate.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-route-convergence-release-gate.zip",
          "byte_size": 459877,
          "sha256": "22a63fe814c3072aef7c6efa3a0f345fc2f8ad92a6c872120727f29bf59b5e44",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "tile-cache-coherence-gate",
          "filename": "greatruth-tb4-tile-cache-coherence-gate.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-tb4-tile-cache-coherence-gate.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-tb4-tile-cache-coherence-gate.zip",
          "byte_size": 264311,
          "sha256": "7c31e70013b6a97acf0657c5f9893d9eaec4e0db14093d7fe9c16df6de36a92d",
          "encoding": "binary",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-terminalbench4-samples.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-terminalbench4-samples.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-terminalbench4-samples.zip",
        "byte_size": 4355377,
        "sha256": "8f65f8d6c1c3c4179f492e7f32a4859a5bcd89553a7e4e751d6fbb4c12b38fdc",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-deepswe-001",
      "legacy_id": "deepswe",
      "slug": "deepswe-zod-v16",
      "title": "DeepSWE · Repo-backed Harbor 任务",
      "title_en": "DeepSWE Repo-backed Harbor Tasks",
      "eyebrow": "DEEPSWE · HARBOR SWE",
      "eyebrow_en": "DEEPSWE · HARBOR SWE",
      "description": "可执行 repo-backed SWE 环境覆盖 Python、Go、Rust 项目的功能修改任务。每个包提供完整题面、Docker 环境、独立测试契约、参考解和来源轨迹记录，评测口径保持按任务独立。",
      "description_en": "Executable repo-backed SWE environments cover feature tasks in Python, Go, and Rust projects. Each package provides a complete instruction, Docker environment, independent test contract, reference solution, and source trajectory record under its own evaluation protocol.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "SWE-bench / Terminal-Bench 风格",
        "label_en": "SWE-bench / Terminal-Bench style",
        "scope": "任务形态对标，非官方子集",
        "scope_en": "Task-format reference, not an official subset"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.08.15",
      "release_date": "2026-08-15",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "verifier",
        "reference_solution",
        "trajectory_selection",
        "rollout_evidence"
      ],
      "tags": [
        "DeepSWE",
        "Harbor",
        "SWE",
        "RLVR",
        "Verifier"
      ],
      "viewer_url": "/deepswe/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "harbor_sub_packages",
        "label": "Harbor 任务包",
        "label_en": "Harbor task packages",
        "display": "8 子包",
        "display_en": "8 packages",
        "basis": "Eight current repo-backed SWE source archives",
        "status_label": "环境集合",
        "status_label_en": "Environment collection"
      },
      "preview": {
        "sequence": [
          "task",
          "tool",
          "verify",
          "reward",
          "output"
        ],
        "metrics": [
          {
            "label": "子包",
            "label_en": "Packages",
            "value": "9"
          },
          {
            "label": "任务形态",
            "label_en": "Task format",
            "value": "Repo SWE"
          },
          {
            "label": "运行边界",
            "label_en": "Runtime boundary",
            "value": "按包声明",
            "value_en": "Per package"
          }
        ],
        "compare": [],
        "note": "每个子包分别保留来源仓库、测试契约和轨迹记录。",
        "note_en": "Each package retains its source repository, test contract, and trajectory record."
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "environment reproducibility",
          "independent verifier boundary",
          "source trajectory record"
        ]
      },
      "record_count": 9,
      "record_unit": "Harbor packages",
      "sample_count": 9,
      "step_count": 0,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Curated Harbor sample bundle",
          "filename": "deepswe-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/deepswe-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/deepswe-sample.json",
          "byte_size": 136691,
          "sha256": "84dd01c85cd03b8c597e2d9320ddd787ee7f857d2282c971e07cdc6c8960c94d",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "Ariadne Codegen Harbor ZIP",
          "filename": "greatruth-deepswe-ariadne-codegen-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-ariadne-codegen-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-ariadne-codegen-harbor-sample.zip",
          "byte_size": 1227094,
          "sha256": "25ed77ea6d6f65afbf0275491c606fc355a5b4ee9be07dd2ad17f5d05e748932",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "Betterleaks Harbor ZIP",
          "filename": "greatruth-deepswe-betterleaks-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-betterleaks-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-betterleaks-harbor-sample.zip",
          "byte_size": 556566,
          "sha256": "a965a2a62af4e0fc64155e2875719b475f9663efce2b8ddc836f50faa1d14b3c",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "envconsul Harbor ZIP",
          "filename": "greatruth-deepswe-envconsul-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-envconsul-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-envconsul-harbor-sample.zip",
          "byte_size": 626011,
          "sha256": "0e919b925b1cf932f0f67d6a557ef589a9d6107480fca643df63c9e074aaaba6",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "fx Harbor ZIP",
          "filename": "greatruth-deepswe-fx-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-fx-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-fx-harbor-sample.zip",
          "byte_size": 403126,
          "sha256": "bff85aef80ee8a1adcb15cb31dbc87b95301360c36c4741cdb4a6912793b6daa",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "lf Harbor ZIP",
          "filename": "greatruth-deepswe-lf-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-lf-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-lf-harbor-sample.zip",
          "byte_size": 149986,
          "sha256": "6800c73dbf79f26f1b6ef95b32b37346b0967801b1dd7f22432e4996c8c04ced",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "Norminette Harbor ZIP",
          "filename": "greatruth-deepswe-norminette-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-norminette-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-norminette-harbor-sample.zip",
          "byte_size": 668294,
          "sha256": "0e3b1249eda0919482d876856ba3bf6849463ddabc1f5329a616f35e1b45267c",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "rel Harbor ZIP",
          "filename": "greatruth-deepswe-rel-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-rel-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-rel-harbor-sample.zip",
          "byte_size": 632454,
          "sha256": "2a7bfcd7d356eda5af48ee5fb8f1214647111223271c3aad15be2d1b32c85ce0",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "ttymap Harbor ZIP",
          "filename": "greatruth-deepswe-ttymap-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-ttymap-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-ttymap-harbor-sample.zip",
          "byte_size": 967903,
          "sha256": "30c505665260e583ab03f5afcc6eddb5fc0cef9e7551239f034f41e05adff394",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "hotpdf Harbor ZIP",
          "filename": "greatruth-deepswe-hotpdf-harbor-sample.zip",
          "media_type": "application/zip",
          "url": "/downloads/greatruth-deepswe-hotpdf-harbor-sample.zip",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-hotpdf-harbor-sample.zip",
          "byte_size": 434147,
          "sha256": "6e5cd255ad766cc379aad9d45c4ba79ded7e9a0609d1dd2a4a591e5586a27f83",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "DeepSWE 五仓库说明书",
          "filename": "greatruth-deepswe-five-repos-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/greatruth-deepswe-five-repos-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-five-repos-guide.pdf",
          "byte_size": 637837,
          "sha256": "3ba1c5daa5bf12612145b7990f4e197260c2a71be55f0ff54e496db99601cfd2",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "hotpdf 说明书",
          "filename": "greatruth-deepswe-hotpdf-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/greatruth-deepswe-hotpdf-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-hotpdf-guide.pdf",
          "byte_size": 160557,
          "sha256": "18106c8f1ab090642a4c94692b0db5bda04e0cca1be2ea2e2a40609086ef877c",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-deepswe-harbor-samples.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-deepswe-harbor-samples.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-deepswe-harbor-samples.zip",
        "byte_size": 5667063,
        "sha256": "d3d361439c52e3ad0ea58f4c57221bceca522d3c14c8b355fea79944c64e2488",
        "encoding": "binary",
        "requires_authentication": false
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "当前 repo-backed SWE 任务覆盖客户端生成、镜像扫描、配置冲突、JSON 导航、命令行补全、本地化、查询预加载和 HTTP 瓦片源。",
        "excerpt_en": "The current repo-backed SWE tasks cover client generation, image scanning, configuration collisions, JSON navigation, command-line completion, localization, query preloads, and HTTP tile sources.",
        "inventory": [
          "当前 Harbor / NewSWE 原始任务包",
          "每包包含题面、环境 Dockerfile、测试配置、参考补丁和轨迹材料",
          "标准 Harbor 包保留 selection-manifest；NewSWE 包保留任务定稿记录",
          "直接下载链接逐包对应来源 ZIP，汇总包只包含这九个文件"
        ],
        "inventory_en": [
          "Current Harbor / NewSWE source task packages",
          "Each package includes an instruction, environment Dockerfile, test configuration, reference patch, and trajectory material",
          "Standard Harbor packages retain selection manifests; NewSWE packages retain task finalization records",
          "Each direct download maps to its source ZIP, and the collection bundle contains only these nine files"
        ]
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-qna-001",
      "legacy_id": "qna",
      "slug": "qna-zod-release-audit",
      "title": "QnA · Zod 契约审计",
      "title_en": "QnA Zod Release Contract Audit",
      "eyebrow": "SWE-ATLAS · QnA",
      "description": "SWE-Atlas 格式的 Zod 只读仓库审计任务，围绕运行时回退、JSON Schema 投影、Codec 与错误拓扑形成可复现证据，并给出下游集成建议。",
      "description_en": "A read-only Zod repository audit in SWE-Atlas format. The task requires reproducible evidence for runtime fallbacks, JSON Schema projections, codecs, and error topology, followed by downstream integration guidance.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "SWE-bench 类仓库任务",
        "label_en": "SWE-bench-style repository task",
        "scope": "只读审计变体",
        "scope_en": "Read-only audit variant"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.07.31",
      "release_date": "2026-07-31",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "repository_snapshot",
        "runtime_probe",
        "rubric",
        "reference_answer",
        "evaluator"
      ],
      "tags": [
        "QnA",
        "SWE-Atlas",
        "Audit",
        "Zod"
      ],
      "viewer_url": "/qna/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "task_contract",
        "label": "任务契约",
        "label_en": "Task contract",
        "display": "完整",
        "display_en": "Complete",
        "basis": "instruction, environment, HLI rubrics and reference answer",
        "status_label": "SWE-Atlas"
      },
      "preview": {
        "sequence": [
          "task",
          "probe",
          "source",
          "answer",
          "rubric"
        ],
        "metrics": [
          {
            "label": "任务结构",
            "label_en": "Task structure",
            "value": "SWE-Atlas"
          },
          {
            "label": "运行方式",
            "label_en": "Execution mode",
            "value": "只读审计",
            "value_en": "Read-only audit"
          },
          {
            "label": "评测",
            "label_en": "Evaluation",
            "value": "HLI Rubric"
          }
        ],
        "note": "任务覆盖四组运行时与源码审计问题，要求提供复现实验证据和四类下游集成建议。",
        "note_en": "The task spans four runtime and source-audit scenarios, requiring reproducible evidence and recommendations for four downstream integration paths."
      },
      "evaluation": {
        "recommended_checks": [
          "runtime evidence completeness",
          "repository symbol references",
          "positive and negative rubric coverage",
          "reference answer and evaluator separation"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Curated Harbor sample bundle",
          "filename": "qna-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/qna-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/qna-sample.json",
          "byte_size": 58827,
          "sha256": "973c4bbb2fed38fff45260375d35911297492dd41f9f26afb49a97984a74e96d",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-qna-zod-release-contract-audit.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-qna-zod-release-contract-audit.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-qna-zod-release-contract-audit.zip",
        "byte_size": 20940,
        "sha256": "03c92386b60072f0754e31b979f89e7cc9e56504a7b56d72306ca9c35b91d8fe",
        "encoding": "binary",
        "requires_authentication": false
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "在固定 Zod 提交上开展只读发布契约审计，通过临时探针、源码符号和关键输出复核 defaults、prefaults、Codec、JSON Schema 与错误路径。",
        "excerpt_en": "Conduct a read-only release-contract audit against a fixed Zod commit. Temporary probes, source symbols, and key outputs are used to verify defaults, prefaults, codecs, JSON Schema behavior, and error paths.",
        "inventory": [
          "SWE-Atlas task.toml 与固定 Zod repository commit",
          "四组场景：运行时回退、JSON Schema、Codec、错误拓扑",
          "正向与负向 HLI Rubric、完整参考答案和评测提示",
          "Node 22 环境中只读分析，临时探针结束后清理"
        ],
        "inventory_en": [
          "SWE-Atlas task.toml and a fixed Zod repository commit",
          "Four scenario groups: runtime fallbacks, JSON Schema, codecs, and error topology",
          "Positive and negative HLI rubrics, complete reference answer, and evaluator prompts",
          "Read-only analysis in Node 22 with temporary probes removed after use"
        ]
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-simulator-001",
      "legacy_id": "simulator",
      "slug": "simulator-m1-m15",
      "title": "Simulator · M1–M15",
      "title_en": "Harbor Science & Engineering Simulators",
      "eyebrow": "SIMULATOR · 15 题",
      "eyebrow_en": "SIMULATOR · 15 TASKS",
      "description": "15 道接真实仿真引擎的策略题。普通 starter 约 0.35 分，oracle 策略为 1.0；独立 verifier 复核时 oracle 全部过线。",
      "description_en": "Fifteen policy tasks connected to real simulation engines. A typical starter scores about 0.35 and the oracle policy scores 1.0; all oracle runs pass independent verifier checks.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "可执行控制 / 优化任务",
        "label_en": "Executable control / optimization tasks",
        "scope": "RLVR 任务形态",
        "scope_en": "RLVR task format"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.07.28",
      "release_date": "2026-07-28",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "simulator",
        "policy",
        "reward",
        "oracle"
      ],
      "tags": [
        "Simulator",
        "Harbor",
        "Continuous Reward",
        "Oracle"
      ],
      "viewer_url": "/simulator/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "task_count",
        "label": "题目数",
        "label_en": "Tasks",
        "score": 15,
        "scale": 15,
        "display": "15 题",
        "display_en": "15 tasks",
        "basis": "M1–M15 package",
        "status_label": "样例包",
        "status_label_en": "Sample package"
      },
      "preview": {
        "sequence": [
          "task",
          "action",
          "state",
          "reward",
          "oracle"
        ],
        "metrics": [
          {
            "label": "题目",
            "label_en": "Tasks",
            "value": "15"
          },
          {
            "label": "starter",
            "value": "≈0.35"
          },
          {
            "label": "oracle",
            "value": "1.0"
          }
        ],
        "compare": [
          {
            "label": "典型梯度",
            "label_en": "Typical reward gradient",
            "left": "starter 0.35",
            "right": "oracle 1.0",
            "left_pct": 35,
            "right_pct": 100
          }
        ],
        "note": "含 PyBaMM：Harbor → E2B 路径上 oracle = 1.0",
        "note_en": "Includes PyBaMM, with oracle = 1.0 on the Harbor to E2B execution path."
      },
      "evaluation": {
        "recommended_checks": [
          "reward gradient",
          "oracle reproducibility",
          "phase trajectory",
          "deterministic replay"
        ]
      },
      "record_count": 15,
      "record_unit": "tasks",
      "sample_count": 15,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Curated Harbor sample bundle",
          "filename": "simulator-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/simulator-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/simulator-sample.json",
          "byte_size": 166216,
          "sha256": "194e579711fc5eda8244a39de1b620e3badbb68db53faed03c9e6abc85fd67af",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-simulator-m1-m15-sample.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-simulator-m1-m15-sample.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-simulator-m1-m15-sample.zip",
        "byte_size": 12374339,
        "sha256": "2db7460cb5328ed95dce6ef253753fce954903677b5187e2787a74d6b2f7d8b6",
        "encoding": "binary",
        "requires_authentication": false
      },
      "content": {
        "label": "题集概览",
        "label_en": "Task-set overview",
        "excerpt": "15 道仿真策略题，各自连接真实引擎与连续 reward。例如晶圆排程、缓存预取、队列控制、SQLite 索引调优等。",
        "excerpt_en": "Fifteen simulation-policy tasks, each connected to a real engine and continuous reward. Scenarios include wafer scheduling, cache prefetching, queue control, and SQLite index tuning.",
        "inventory": [
          "15 题（14 道 legacy + 1 道 PyBaMM）",
          "每题含 instruction、simulator、policy 接口与 oracle",
          "starter 约 0.35，oracle 为 1.0；separate-verifier 上 oracle 14/14 过线"
        ],
        "inventory_en": [
          "15 tasks: 14 legacy tasks and 1 PyBaMM task",
          "Each task includes an instruction, simulator, policy interface, and oracle",
          "Starter score about 0.35 and oracle score 1.0; oracle passes 14/14 separate-verifier checks"
        ]
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-gdpevo-001",
      "legacy_id": "gdpevo",
      "slug": "gdpevo-enterprise-workflows",
      "title": "GDPevo · 5 垂域",
      "title_en": "GDPevo Enterprise Workflows",
      "eyebrow": "GDPEVO · 企业工作流",
      "eyebrow_en": "GDPEVO · ENTERPRISE WORKFLOWS",
      "description": "五个企业垂域办公环境，每包含训练/测试题与规则评分器。公开成绩组级 best 约 47–75%；上游未放出原始 rollout。",
      "description_en": "Five enterprise workflow environments, each with train and test tasks plus a rule-based evaluator. Published group-best scores range from about 47% to 75%; raw rollouts are not included upstream.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "τ-bench 类",
        "label_en": "τ-bench style",
        "scope": "多工具状态化工作流",
        "scope_en": "Stateful multi-tool workflows"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.07.27",
      "release_date": "2026-07-27",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "business_state",
        "tool_api",
        "policy",
        "gold_answer",
        "rule_evaluator"
      ],
      "tags": [
        "GDPevo",
        "CRM",
        "Finance",
        "Workflow"
      ],
      "viewer_url": "/gdpevo/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "group_best",
        "label": "组级 best",
        "label_en": "Group best",
        "score": 0.47,
        "scale": 1,
        "display": "47–75%",
        "basis": "GPT-5.5/Codex group-level best range",
        "status_label": "上游实测",
        "status_label_en": "Upstream evaluation"
      },
      "preview": {
        "sequence": [
          "brief",
          "retrieve",
          "decide",
          "write",
          "score"
        ],
        "metrics": [
          {
            "label": "垂域",
            "label_en": "Domains",
            "value": "5"
          },
          {
            "label": "组级 best",
            "label_en": "Group best",
            "value": "47–75%"
          },
          {
            "label": "可变题",
            "label_en": "Mutable tasks",
            "value": "1"
          }
        ],
        "note": "Gold 经 selected_task/eval.sh 可拿满分；勿伪造轨迹",
        "note_en": "Gold reaches full credit under selected_task/eval.sh; trajectories must remain authentic."
      },
      "evaluation": {
        "recommended_checks": [
          "gold schema",
          "rule evaluator score",
          "policy adherence",
          "cross-system consistency"
        ]
      },
      "record_count": 5,
      "record_unit": "scenarios",
      "sample_count": 5,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Curated Harbor sample bundle",
          "filename": "gdpevo-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/gdpevo-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/gdpevo-sample.json",
          "byte_size": 54219,
          "sha256": "2cec5e94fbfe97adc50b9fec76a33b4f9903dfd72f712bddea62de4feeaa3078",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-gdpevo-sample.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-gdpevo-sample.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-gdpevo-sample.zip",
        "byte_size": 1320742,
        "sha256": "7189d0886c8e8e9b87b5abdc519cee9c7d1bcc8bdab5e9eba5d487d3adc91c29",
        "encoding": "binary",
        "requires_authentication": false
      },
      "content": {
        "label": "垂域任务",
        "label_en": "Domain tasks",
        "excerpt": "五个企业工作流：CRM 活动销售与财务交接、电商月末关账、银行信贷委员会、医疗支付分诊、M&A 交割决策。",
        "excerpt_en": "Five enterprise workflows: CRM campaign handoff to sales and finance, e-commerce inventory correction and month-end close, bank credit-committee decisions, healthcare payer operations triage, and M&A transaction review and closing.",
        "inventory": [
          "每垂域 1 个环境 + 5 train + 5 test",
          "提交结构化 JSON，由规则评分器打分",
          "公开组级 best 约 47–75%（GPT-5.5 / Codex）；不含 raw rollout"
        ],
        "inventory_en": [
          "Per domain: 1 environment, 5 train tasks, and 5 test tasks",
          "Structured JSON submissions scored by rule-based evaluators",
          "Published group-best range about 47% to 75% for GPT-5.5 / Codex; raw rollouts not included"
        ]
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-cyber-001",
      "legacy_id": "cyber",
      "slug": "cyber-defense-seed176",
      "title": "Cyber Defense · seed 176",
      "title_en": "Cyber Defense Seed 176",
      "eyebrow": "CYBER · SEED 176",
      "description": "开放式 SOC 日志狩猎：约 15.5 万条 Windows/Sysmon 事件，自写 SQL（最多 50 次）并提交恶意时间戳。本样例包含 GPT-5.5 baseline 轨迹（71 轮，outcome=gave_up）。",
      "description_en": "An open-ended SOC log-hunting environment with about 155,000 Windows and Sysmon events. The agent writes up to 50 SQL queries and submits malicious timestamps. This sample includes a 71-turn GPT-5.5 baseline rollout with outcome=gave_up.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "Cybench 类",
        "label_en": "Cybench style",
        "scope": "安全调查与工具执行",
        "scope_en": "Security investigation and tool execution"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.07",
      "release_date": "2026-07-28",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "event_logs",
        "sql_tool",
        "timestamps",
        "rollout"
      ],
      "tags": [
        "Cyber",
        "SOC",
        "Threat Hunting",
        "Harbor"
      ],
      "viewer_url": "/cyber/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "baseline_rollout",
        "label": "Baseline",
        "display": "gave_up",
        "basis": "gpt-5.5 baseline · audit v14 coverage 0.764",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "sequence": [
          "brief",
          "sql",
          "pivot",
          "submit",
          "score"
        ],
        "metrics": [
          {
            "label": "Turns",
            "value": "71"
          },
          {
            "label": "SQL",
            "value": "50"
          },
          {
            "label": "Submitted",
            "value": "147"
          }
        ],
        "note": "同 seed 的 v14_natural 覆盖约 0.764；详见审计页",
        "note_en": "For the same seed, v14_natural reaches approximately 0.764 coverage; see the audit page for details."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "调查可疑入侵：基于 Windows 主机事件日志切片，找出属于攻击链的事件时间戳，并原样提交 ISO-8601 时间戳。",
        "excerpt_en": "Investigate a suspected intrusion using a slice of Windows host event logs, identify timestamps belonging to the attack chain, and submit the exact ISO-8601 timestamps.",
        "inventory": [
          "约 15.5 万条事件日志 + SQL schema",
          "查询预算 50 次，每次最多返回 10 行",
          "含 GPT-5.5 baseline Harbor 轨迹（71 轮 / 50 SQL）",
          "审计页另记同 seed 的 v14_natural 覆盖 0.764"
        ],
        "inventory_en": [
          "About 155,000 event records plus the SQL schema",
          "A 50-query budget with at most 10 returned rows per query",
          "GPT-5.5 baseline Harbor rollout with 71 turns and 50 SQL calls",
          "The audit page also records 0.764 coverage for v14_natural on the same seed"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "timestamp fidelity",
          "sql budget",
          "coverage vs labels"
        ]
      },
      "record_count": 1,
      "record_unit": "trajectories",
      "sample_count": 1,
      "product_line": "rl_sft_environment",
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Curated Harbor sample bundle",
          "filename": "cyber-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/cyber-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/cyber-sample.json",
          "byte_size": 50021,
          "sha256": "635517c8b54eea5104f5dce786b06092f0978199b9aa2c4fad88a2e2b6bbb437",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "bundle",
          "title": "Harbor task and rollout bundle",
          "filename": "greatruth-cyber-defense-seed176.tar.gz",
          "media_type": "application/gzip",
          "url": "/downloads/greatruth-cyber-defense-seed176.tar.gz",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-cyber-defense-seed176.tar.gz",
          "byte_size": 11621003,
          "sha256": "7b2751fbc6274dc0b09ebce543ce6fb92c3ae3baf51f6369e084c8e3a482d724",
          "encoding": "binary",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "精简样例 JSON",
          "filename": "greatruth-cyber-defense-seed176-sample.json",
          "media_type": "application/json",
          "url": "/downloads/greatruth-cyber-defense-seed176-sample.json",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-cyber-defense-seed176-sample.json",
          "byte_size": 48204,
          "sha256": "2c3f5079e605275e0a2e55f2d9a638a0898ae89391c7b5d40db1a5bd53391e43",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-cyber-defense-seed176.tar.gz",
        "media_type": "application/gzip",
        "url": "/downloads/greatruth-cyber-defense-seed176.tar.gz",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-cyber-defense-seed176.tar.gz",
        "byte_size": 11621003,
        "sha256": "7b2751fbc6274dc0b09ebce543ce6fb92c3ae3baf51f6369e084c8e3a482d724",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-env-cyber-rf-vulhub-001",
      "slug": "cyber-rf-vulhub",
      "title": "Cyber · RF-VULHUB 双层靶场",
      "title_en": "Cyber · RF-VULHUB Two-layer Range",
      "eyebrow": "CYBER · RF-VULHUB",
      "eyebrow_en": "CYBER · RF-VULHUB",
      "description": "隔离 Docker 靶场中的双层四漏洞任务，包含完整题面、离线镜像、独立 Verifier、来源自测、修复版本对照与说明书。",
      "description_en": "A two-layer, four-vulnerability task in an isolated Docker range, with complete instructions, offline images, an independent verifier, source self-tests, patched-version controls, and a guide.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "Vulhub 靶场",
        "label_en": "Vulhub range",
        "scope": "本地隔离双层任务",
        "scope_en": "Local isolated two-layer task"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.14",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "en"
      ],
      "modalities": [
        "instruction",
        "docker_environment",
        "verifier",
        "validation"
      ],
      "tags": [
        "Cyber",
        "Vulhub",
        "Docker",
        "Verifier"
      ],
      "viewer_url": "/cyber-rf-vulhub/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_self_test",
        "label": "来源自测",
        "label_en": "Source self-test",
        "display": "任务契约",
        "display_en": "Task contracts",
        "basis": "Single-task source validation; Elasticsearch patched-image control remains untested",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "TASK CONTRACT",
        "label_en": "TASK CONTRACT",
        "summary": "四节点有序核验",
        "summary_en": "Four-node ordered verification",
        "sequence": [
          "task",
          "range",
          "verifier"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "1"
          }
        ],
        "note": "无模型轨迹或批次通过率；修复版本对照为部分完成。",
        "note_en": "No model traces or batch pass rates; patched-version controls are partial."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "隔离 Docker 靶场中的双层四漏洞任务，包含完整题面、离线镜像、独立 Verifier、来源自测、修复版本对照与说明书。",
        "excerpt_en": "A two-layer, four-vulnerability task in an isolated Docker range, with complete instructions, offline images, an independent verifier, source self-tests, patched-version controls, and a guide.",
        "inventory": [
          "完整题面与变体",
          "隔离环境及五份离线镜像",
          "Verifier、自测、修复版本对照与 PDF"
        ],
        "inventory_en": [
          "Complete instruction and variants",
          "Isolated environment and five offline images",
          "Verifier, self-tests, patched-version controls, and PDF"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "ordered prefix scoring",
          "exact token verification",
          "network isolation",
          "patched-version boundaries"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "RF-VULHUB structured sample",
          "filename": "cyber-rf-vulhub-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/cyber-rf-vulhub-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/cyber-rf-vulhub-sample.json",
          "byte_size": 26998,
          "sha256": "84f689b785d37faf5df8fcf7353bb72fa852eb624cfef04f3c859f7f119eb594",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "说明书 PDF",
          "filename": "greatruth-cyber-rf-vulhub-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/greatruth-cyber-rf-vulhub-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-cyber-rf-vulhub-guide.pdf",
          "byte_size": 110653,
          "sha256": "ff34718c95605114902b3d7e6abb4eefbe4cf7af362ba52b7f38dfaea8f6d064",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-cyber-rf-vulhub-four-v2l.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-cyber-rf-vulhub-four-v2l.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-cyber-rf-vulhub-four-v2l.zip",
        "byte_size": 1159705294,
        "sha256": "faddfcc1534b4ce957b7e60e99937412bf2e485bb241971223751940f2753c9a",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-clawbench-001",
      "slug": "clawbench-browser-tasks",
      "title": "ClawBench · 浏览器多工具任务",
      "title_en": "ClawBench · Browser Tool-use Tasks",
      "eyebrow": "CLAWBENCH · BROWSER RL",
      "eyebrow_en": "CLAWBENCH · BROWSER RL",
      "description": "浏览器交互环境覆盖表单提交、审批、查询与工单操作；每项任务保留任务契约、附件、独立请求核验和 Pass@8 校准记录。",
      "description_en": "A browser-interaction environment covering form submission, approvals, lookups, and support workflows. Each task retains its contract, attachments, independent request checks, and Pass@8 calibration record.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "ClawBench 风格",
        "label_en": "ClawBench style",
        "scope": "浏览器多工具任务，私有合成样例",
        "scope_en": "Browser tool-use tasks, private synthetic sample"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.07",
      "release_date": "2026-09-07",
      "status": "verified",
      "languages": [
        "en",
        "zh-CN"
      ],
      "modalities": [
        "instruction",
        "browser_environment",
        "request_interception",
        "semantic_judge",
        "pass8_calibration"
      ],
      "tags": [
        "ClawBench",
        "Browser",
        "Tool Use",
        "RLVR",
        "Synthetic"
      ],
      "viewer_url": "/clawbench/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "pass8_calibration",
        "label": "Pass@8 校准",
        "label_en": "Pass@8 calibration",
        "display": "任务契约",
        "display_en": "Task contracts",
        "basis": "10 private synthetic tasks with complete eight-valid-rollout calibration",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "TASK CONTRACT",
        "label_en": "TASK CONTRACT",
        "summary": "浏览器请求核验",
        "summary_en": "Browser request checks",
        "sequence": [
          "task",
          "browser",
          "request",
          "judge",
          "pass8"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "10"
          },
          {
            "label": "请求形态",
            "label_en": "Request shapes",
            "value": "9"
          },
          {
            "label": "动作类型",
            "label_en": "Action types",
            "value": "8"
          }
        ],
        "note": "每项任务独立记录 Pass@8 与请求语义核验。",
        "note_en": "Each task retains its own Pass@8 and semantic request checks."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "十项私有合成浏览器任务，覆盖结构化表单、审批、查询和支持工作流。",
        "excerpt_en": "Ten private synthetic browser tasks spanning structured forms, approvals, lookups, and support workflows.",
        "inventory": [
          "完整任务契约与 extra_info 附件",
          "拦截请求、参数字段和语义 Judge",
          "每项任务八次有效 rollout 的 Pass@8 记录"
        ],
        "inventory_en": [
          "Complete task contracts and extra_info attachments",
          "Intercepted requests, parameter fields, and semantic judging",
          "Pass@8 records from eight valid rollouts per task"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "task contract completeness",
          "request interception and semantic fields",
          "valid rollout denominator",
          "attachment availability"
        ]
      },
      "record_count": 10,
      "record_unit": "tasks",
      "sample_count": 10,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "ClawBench structured sample",
          "filename": "clawbench-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/clawbench-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/clawbench-sample.json",
          "byte_size": 68719,
          "sha256": "7983b6ab6100ba4fe64c3b05305bb29990997db3c41d7edcfcc6d909b8114a60",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "配套说明书 PDF",
          "filename": "greatruth-clawbench-diverse-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/greatruth-clawbench-diverse-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-clawbench-diverse-guide.pdf",
          "byte_size": 590715,
          "sha256": "fecc7d11af0bd9c98078ecddf634fec5c8c752e51d882795f2827d29af9a48fe",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-clawbench-diverse-sample.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-clawbench-diverse-sample.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-clawbench-diverse-sample.zip",
        "byte_size": 212148,
        "sha256": "ea019d0bc3db80d65965ab31ddba95dadf87ecd7cd860e7611a31f3eeec77ef2",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-clawbench-wps-001",
      "slug": "clawbench-wps",
      "title": "OpenClawBench · WPS 桌面任务",
      "title_en": "OpenClawBench · WPS Desktop Task",
      "eyebrow": "OPENCLAWBENCH · WPS",
      "eyebrow_en": "OPENCLAWBENCH · WPS",
      "description": "在断网 WPS 桌面环境中制作宏观排名工作簿，保留完整题面、文件核验和截图轨迹。",
      "description_en": "Create a macroeconomic ranking workbook in an offline WPS desktop, with complete instructions, file verification, and screenshot trajectories.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "ClawBench 风格",
        "label_en": "ClawBench style",
        "scope": "原生 WPS 桌面工作簿任务",
        "scope_en": "Native WPS desktop workbook task"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.11",
      "release_date": "2026-09-11",
      "status": "verified",
      "languages": [
        "en",
        "zh-CN"
      ],
      "modalities": [
        "instruction",
        "desktop_environment",
        "file_verifier",
        "screenshot_trajectory"
      ],
      "tags": [
        "OpenClawBench",
        "WPS",
        "Desktop",
        "RLVR"
      ],
      "viewer_url": "/clawbench-wps/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_verifier",
        "label": "文件核验",
        "label_en": "File verification",
        "display": "任务契约",
        "display_en": "Task contracts",
        "basis": "Source run status and workbook checks",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "TASK CONTRACT",
        "label_en": "TASK CONTRACT",
        "summary": "WPS 工作簿文件核验",
        "summary_en": "WPS workbook file checks",
        "sequence": [
          "task",
          "desktop",
          "xlsx",
          "verifier"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "1"
          }
        ],
        "note": "运行结果沿用来源记录。",
        "note_en": "Run outcomes follow source records."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "在断网 WPS 桌面环境中制作宏观排名工作簿，保留完整题面、文件核验和截图轨迹。",
        "excerpt_en": "Create a macroeconomic ranking workbook in an offline WPS desktop, with complete instructions, file verification, and screenshot trajectories.",
        "inventory": [
          "完整题面与 CSV",
          "WPS 环境与文件核验",
          "截图轨迹与提交文件"
        ],
        "inventory_en": [
          "Complete instruction and CSV",
          "WPS environment and file verifier",
          "Screenshot trajectories and submissions"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "workbook checks",
          "source run outcomes"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "WPS structured sample",
          "filename": "clawbench-wps-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/clawbench-wps-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/clawbench-wps-sample.json",
          "byte_size": 13156,
          "sha256": "71035f1b768447ea95e11b9029c3b6ce3ca1a7e99ea5702019290989e00fe68a",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "companion_sample",
          "title": "配套说明书 PDF",
          "filename": "greatruth-clawbench-wps-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/greatruth-clawbench-wps-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/greatruth-clawbench-wps-guide.pdf",
          "byte_size": 242079,
          "sha256": "a1403c996dd95b4d5fb45132320b4f4dbb234b22959875f725c0585ace189ce6",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-clawbench-wps-standalone.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-clawbench-wps-standalone.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-clawbench-wps-standalone.zip",
        "byte_size": 73293735,
        "sha256": "238c8ca0365e112822a056d51342190b787190ca8db53d10a2a6cc2461cac597",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-computer-use-001",
      "slug": "computer-use-os-sandboxes",
      "title": "Computer Use · OS 沙箱任务",
      "title_en": "Computer Use · OS Sandbox Tasks",
      "eyebrow": "COMPUTER USE · DESKTOP RL",
      "eyebrow_en": "COMPUTER USE · DESKTOP RL",
      "description": "带桌面环境、截图观察、鼠标交互和文件状态 Verifier 的 Computer Use 任务样例；每项任务保留沙箱材料与 Pass@8 运行记录。",
      "description_en": "Computer Use tasks with a desktop environment, screenshot observations, mouse interactions, and file-state verifiers. Each task retains sandbox materials and Pass@8 run records.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "OSWorld / BrowserGym 风格",
        "label_en": "OSWorld / BrowserGym style",
        "scope": "桌面交互任务，原创本地工作流",
        "scope_en": "Desktop interaction, original local workflows"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.07",
      "release_date": "2026-09-07",
      "status": "verified",
      "languages": [
        "en",
        "zh-CN"
      ],
      "modalities": [
        "instruction",
        "desktop_environment",
        "screenshot_observation",
        "mouse_interaction",
        "file_state_verifier"
      ],
      "tags": [
        "Computer Use",
        "Desktop",
        "Browser",
        "RLVR",
        "Screenshot"
      ],
      "viewer_url": "/computer-use/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "pass8_calibration",
        "label": "Pass@8 校准",
        "label_en": "Pass@8 calibration",
        "display": "文件状态",
        "display_en": "File-state checks",
        "basis": "Two L4 desktop tasks with screenshot evidence and cleanup records",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "DESKTOP CONTRACT",
        "label_en": "DESKTOP CONTRACT",
        "summary": "截图与文件核验",
        "summary_en": "Screenshot and file checks",
        "sequence": [
          "task",
          "screen",
          "mouse",
          "artifact",
          "verify"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "2"
          },
          {
            "label": "运行环境",
            "label_en": "Runtime",
            "value": "E2B desktop"
          },
          {
            "label": "观察方式",
            "label_en": "Observation",
            "value": "Screenshot"
          }
        ],
        "note": "Pass@8 与不可用 slot 按任务分别保留。",
        "note_en": "Pass@8 and unavailable slots are retained per task."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "两项 L4 浏览器桌面任务覆盖地图圈选、拖拽排序、风险确认和文件交付。",
        "excerpt_en": "Two L4 browser-desktop tasks cover map selection, drag ordering, risk confirmation, and file delivery.",
        "inventory": [
          "E2B desktop、浏览器和本地 fixture",
          "截图观察与点击、拖拽等鼠标动作",
          "输出文件、内容匹配和源文件保留检查"
        ],
        "inventory_en": [
          "E2B desktop, browser, and local fixtures",
          "Screenshot observations with click and drag actions",
          "Output, content-match, and source-preservation checks"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "desktop environment reproducibility",
          "screenshot and trajectory completeness",
          "file-state verifier nodes",
          "cleanup confirmation"
        ]
      },
      "record_count": 2,
      "record_unit": "tasks",
      "sample_count": 2,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Computer Use structured sample",
          "filename": "computer-use-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/computer-use-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/computer-use-sample.json",
          "byte_size": 26819,
          "sha256": "77448695c11fe2be0f28e52bb8a3b50c2ce5d75f4af755dc6a9be0af51db5a5a",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-computer-use-os-sandboxes.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-computer-use-os-sandboxes.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-computer-use-os-sandboxes.zip",
        "byte_size": 14495403,
        "sha256": "b841436ab6ba48b0ac923ffdd9ef1811a286f12ef2c7dde9bb3f75c9b0e7c71f",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-automationbench-001",
      "slug": "automationbench-support-workflows",
      "title": "AutomationBench · 客户支持工作流",
      "title_en": "AutomationBench · Support Workflows",
      "eyebrow": "AUTOMATIONBENCH · BUSINESS TOOLS",
      "eyebrow_en": "AUTOMATIONBENCH · BUSINESS TOOLS",
      "description": "七项客户支持自动化任务，覆盖 SLA 升级与客户协同；每项保留离线业务状态、工具接口、断言评测及八次有效运行的来源校准记录。",
      "description_en": "Seven customer-support automation tasks spanning SLA escalation and customer coordination. Each retains offline business state, tool interfaces, assertion evaluation, and source calibration from eight valid trials.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "AutomationBench 风格",
        "label_en": "AutomationBench style",
        "scope": "状态化业务系统多工具任务",
        "scope_en": "Stateful multi-tool business-system tasks"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.25",
      "release_date": "2026-09-25",
      "status": "verified",
      "languages": [
        "en"
      ],
      "modalities": [
        "instruction",
        "business_state",
        "tool_api",
        "assertion_evaluator",
        "pass8_calibration"
      ],
      "tags": [
        "AutomationBench",
        "Support",
        "Business Tools",
        "RLVR",
        "Assertions"
      ],
      "viewer_url": "/automationbench/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_qualification",
        "label": "来源校准",
        "label_en": "Source calibration",
        "display": "7 项合格任务",
        "display_en": "7 qualified tasks",
        "basis": "Seven source tasks with eight valid trials each and strict assertion evaluation",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "AUTOMATION CONTRACT",
        "label_en": "AUTOMATION CONTRACT",
        "summary": "业务状态与断言评测",
        "summary_en": "Business state and assertion evaluation",
        "sequence": [
          "task",
          "state",
          "tools",
          "actions",
          "assertions"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "7"
          },
          {
            "label": "有效运行",
            "label_en": "Valid trials",
            "value": "56"
          },
          {
            "label": "严格通过",
            "label_en": "Strict passes",
            "value": "15"
          }
        ],
        "note": "来源校准按任务解释；本站未重跑模型或评测器。",
        "note_en": "Source calibration is interpreted per task; this site has not rerun models or evaluators."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "客户支持工作流要求智能体在离线业务状态中检索事实、调用工具并满足逐项断言。",
        "excerpt_en": "Customer-support workflows require the agent to retrieve facts, call tools, and satisfy itemized assertions in offline business state.",
        "inventory": [
          "七项完整英文题面与业务初始状态",
          "22–35 个工具声明和 24–33 项断言/任务",
          "每题八次有效运行、严格通过及原创性记录"
        ],
        "inventory_en": [
          "Seven complete English instructions and business initial states",
          "22–35 tool declarations and 24–33 assertions per task",
          "Eight valid trials per task with strict-pass and originality records"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "task contract completeness",
          "tool and assertion counts",
          "valid-trial denominator",
          "source calibration boundaries"
        ]
      },
      "record_count": 7,
      "record_unit": "tasks",
      "sample_count": 7,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "AutomationBench structured sample",
          "filename": "automationbench-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/automationbench-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/automationbench-sample.json",
          "byte_size": 27418,
          "sha256": "34a60f8ad9603a1f70820f9f611efbb5c4c1078ba563b1ff604d5cfbd2c41074",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-automationbench-support-workflows.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-automationbench-support-workflows.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-automationbench-support-workflows.zip",
        "byte_size": 17049190,
        "sha256": "56f17cf91485fe5ff5de294d76b19d8eef87291f092e4d6535144d8a898da71f",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-env-jobbench-001",
      "slug": "jobbench-professional-deliverables",
      "title": "JobBench · 专业交付任务",
      "title_en": "JobBench · Professional Deliverable Tasks",
      "eyebrow": "JOBBENCH · PROFESSIONAL WORK",
      "eyebrow_en": "JOBBENCH · PROFESSIONAL WORK",
      "description": "五项文件密集型专业任务，覆盖财务对账、商业结算、能源权益、调查统计和无障碍发布审计；保留完整输入、参考交付物、Rubric 与八次来源运行。",
      "description_en": "Five file-intensive professional tasks covering financial reconciliation, commercial settlement, energy entitlement, survey statistics, and accessibility release auditing, with complete inputs, reference outputs, rubrics, and eight source runs.",
      "buyer_type": "RL 环境 + Verifier",
      "buyer_type_en": "RL environment + verifier",
      "benchmark_anchor": {
        "label": "JobBench 风格",
        "label_en": "JobBench style",
        "scope": "长程专业文档与表格交付",
        "scope_en": "Long-horizon professional document and spreadsheet delivery"
      },
      "category": "rl_sft_environment",
      "visual_kind": "env_matrix",
      "version": "2026.09.25",
      "release_date": "2026-09-25",
      "status": "verified",
      "languages": [
        "en"
      ],
      "modalities": [
        "instruction",
        "office_files",
        "document_generation",
        "spreadsheet_generation",
        "rubric_evaluator",
        "trajectory"
      ],
      "tags": [
        "JobBench",
        "Professional Work",
        "Office Files",
        "RLVR",
        "Rubric"
      ],
      "viewer_url": "/jobbench/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_qualification",
        "label": "来源资格",
        "label_en": "Source qualification",
        "display": "5 项合格任务",
        "display_en": "5 qualified tasks",
        "basis": "Five source-qualified tasks with eight archived runs each",
        "status_label": "环境样例",
        "status_label_en": "Environment sample"
      },
      "preview": {
        "label": "PROFESSIONAL DELIVERABLE",
        "label_en": "PROFESSIONAL DELIVERABLE",
        "summary": "多文件理解与可审计交付",
        "summary_en": "Multi-file reasoning and auditable delivery",
        "sequence": [
          "brief",
          "sources",
          "analysis",
          "deliverables",
          "rubric"
        ],
        "metrics": [
          {
            "label": "任务",
            "label_en": "Tasks",
            "value": "5"
          },
          {
            "label": "归档运行",
            "label_en": "Archived runs",
            "value": "40"
          },
          {
            "label": "来源均分",
            "label_en": "Source mean",
            "value": "0.470"
          }
        ],
        "note": "分数与 QUALIFIED 决策沿用来源记录，按任务分别解释。",
        "note_en": "Scores and QUALIFIED decisions follow source records and are interpreted per task."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "五项任务要求读取多种办公文件，产出可复核的工作簿、文档、CSV 或 Markdown 交付物。",
        "excerpt_en": "Five tasks require reading varied office files and producing reviewable workbooks, documents, CSV, or Markdown deliverables.",
        "inventory": [
          "五项完整英文任务及全部求解器可见输入",
          "参考交付物、Rubric、计算核验与 Judge 配置",
          "每项八次轨迹、逐次评分和资格记录"
        ],
        "inventory_en": [
          "Five complete English tasks and all solver-visible inputs",
          "Reference outputs, rubrics, calculation checks, and judge configuration",
          "Eight trajectories per task with per-run grades and qualification records"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction and input completeness",
          "deliverable formats",
          "rubric and fixture boundaries",
          "per-task source calibration"
        ]
      },
      "record_count": 5,
      "record_unit": "tasks",
      "sample_count": 5,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "JobBench structured sample",
          "filename": "jobbench-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/harbor/jobbench-sample.json",
          "content_url": "https://display.greatruth.cloud/data/harbor/jobbench-sample.json",
          "byte_size": 37047,
          "sha256": "abf4a006858840e1cba440db9273fbb092156d03f74d1f593488f99a45be0526",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "greatruth-jobbench-professional-tasks.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-jobbench-professional-tasks.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-jobbench-professional-tasks.zip",
        "byte_size": 11184335,
        "sha256": "e12142d672e11ee551a93c8363f9ff262e41850418401a7ff5f41fd67bbd7ef0",
        "encoding": "binary",
        "requires_authentication": false
      },
      "product_line": "rl_sft_environment"
    },
    {
      "id": "gr-synthetic-mcp-001",
      "slug": "mcp-selected-17-traces",
      "title": "MCP 多工具合成轨迹",
      "title_en": "MCP Multi-tool Synthetic Trajectories",
      "eyebrow": "MCP · ENVIRONMENT-AWARE TRAJECTORY",
      "description": "面向复杂业务任务的完整合成轨迹，覆盖任务题面、初始与最终环境、MCP 工具配置、调用过程、自动评测和交付产物。",
      "description_en": "Complete synthetic trajectories for complex business tasks, including task prompts, initial and final environments, MCP tool configuration, call records, automated evaluation, and delivered artifacts.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "Agent Workflow / Tool-use",
        "label_en": "Agent workflow / tool use",
        "scope": "合成任务轨迹",
        "scope_en": "Synthetic task trajectories"
      },
      "category": "high_quality_trajectory",
      "visual_kind": "synthetic_trace",
      "version": "2026.07.29",
      "release_date": "2026-07-29",
      "status": "verified",
      "languages": [
        "en",
        "zh-CN"
      ],
      "modalities": [
        "task_prompt",
        "environment_snapshot",
        "mcp_profile",
        "tool_call",
        "tool_result",
        "workspace_delta",
        "evaluation",
        "artifact"
      ],
      "tags": [
        "MCP",
        "Synthetic",
        "Trajectory",
        "Environment"
      ],
      "viewer_url": "/mcp-traces/viewer.html",
      "schema_url": "/data/schemas/mcp-synthetic-traces.schema.json",
      "quality": {
        "metric": "task_pass_rate",
        "label": "任务通过",
        "label_en": "Tasks passed",
        "status_label": "合成轨迹",
        "status_label_en": "Synthetic trajectories",
        "score": 17,
        "scale": 17,
        "display": "17/17",
        "basis": "307/307 项评测检查通过",
        "basis_en": "307/307 evaluation checks passed"
      },
      "preview": {
        "label": "DATA FLOW",
        "summary": "MCP + ENV",
        "sequence": [
          "prompt",
          "environment",
          "mcp",
          "artifact",
          "evaluation"
        ],
        "metrics": [
          {
            "label": "环境状态",
            "label_en": "Environment state",
            "value": "前 / 后",
            "value_en": "Before / after"
          },
          {
            "label": "工具过程",
            "label_en": "Tool execution",
            "value": "MCP"
          },
          {
            "label": "评测证据",
            "label_en": "Evaluation evidence",
            "value": "Checks"
          }
        ],
        "compare": [
          {
            "label": "MCP 调用结果",
            "label_en": "MCP call outcomes",
            "left": "458 成功",
            "left_en": "458 succeeded",
            "right": "20 失败",
            "right_en": "20 failed",
            "left_pct": 96,
            "right_pct": 4
          }
        ],
        "note": "307/307 项检查通过；18/20 次失败调用完成恢复，2 次仍处于任务允许阈值内。",
        "note_en": "307/307 checks passed. 18/20 failed calls recovered, while 2 remained within the task's allowed threshold."
      },
      "content": {
        "label": "交付结构",
        "label_en": "Delivery structure",
        "excerpt": "每条轨迹形成独立任务交付目录，环境状态、工具过程、模型轨迹、质量检查和产物文件均可逐项复核。",
        "excerpt_en": "Each trajectory is delivered as a self-contained task directory. Environment state, tool execution, model calls, quality checks, and artifact files can be reviewed independently.",
        "inventory": [
          "复杂业务任务与多类 MCP 服务组合",
          "初始与最终环境、状态差异和工作区变化",
          "完整工具调用、失败恢复与自动评测证据",
          "按任务交付模型轨迹、执行记录与产物"
        ],
        "inventory_en": [
          "Complex business tasks spanning multiple MCP services",
          "Initial and final environments, state deltas, and workspace changes",
          "Complete tool calls, failure recovery, and automated evaluation evidence",
          "Per-task model trajectories, execution records, and artifacts"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "environment state completeness",
          "MCP call and result pairing",
          "failure recovery evidence",
          "output contract checks",
          "task-level pass status"
        ]
      },
      "record_count": 17,
      "record_unit": "trajectories",
      "sample_count": 17,
      "product_line": "high_quality_trajectory",
      "download": {
        "filename": "greatruth-mcp-synthetic-trajectories-17.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-mcp-synthetic-trajectories-17.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-mcp-synthetic-trajectories-17.zip",
        "byte_size": 5505790,
        "sha256": "844d41080021102ef9caabc5307a9cef73b1d28b937c9cd47a60efbbb8d21aa0",
        "encoding": "binary",
        "requires_authentication": false
      },
      "step_count": 216,
      "observability": {
        "model_call_count": 216,
        "tool_call_count": 495,
        "mcp_call_count": 478,
        "trace_record_count": 216,
        "configured_mcp_service_count": 12,
        "output_count": 53
      },
      "artifacts": [
        {
          "role": "curated_sample",
          "title": "MCP synthetic trajectory display sample",
          "filename": "mcp-traces-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/mcp-synthetic-traces.schema.json",
          "url": "/data/synthetic/mcp-traces-sample.json",
          "content_url": "https://display.greatruth.cloud/data/synthetic/mcp-traces-sample.json",
          "byte_size": 834392,
          "sha256": "dd7ad0b72427071cc65dc600a66484394ccebb802e4b57589c87b42c329b3843",
          "record_count": 17,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-synthetic-mixed-001",
      "slug": "mixed-collaborative-trajectories",
      "title": "WorkBuddy 多智能体协作轨迹",
      "title_en": "WorkBuddy Multi-agent Trajectories",
      "eyebrow": "WORKBUDDY · TASK-TREE ORCHESTRATION",
      "description": "基于 WorkBuddy 轨迹合成框架的多智能体协作样例。框架以任务树驱动多轮任务演进，在关键节点配置角色化子代理、协作约束与结果合并，并保留 CodeBuddy Code 运行过程和最终工作区。",
      "description_en": "Multi-agent collaboration samples produced with the WorkBuddy trajectory-synthesis framework. WorkBuddy drives multi-round task evolution through a task tree, assigns role-specific subagents and collaboration constraints at selected nodes, merges their results, and retains the CodeBuddy Code execution record and final workspace.",
      "buyer_type": "SFT 轨迹",
      "buyer_type_en": "SFT trajectory",
      "benchmark_anchor": {
        "label": "Multi-agent Workflow",
        "label_en": "Multi-agent workflow",
        "scope": "任务树与协作轨迹",
        "scope_en": "Task-tree collaboration trajectories"
      },
      "category": "high_quality_trajectory",
      "visual_kind": "collaborative_trace",
      "version": "2026.07.23",
      "release_date": "2026-07-23",
      "status": "reviewed",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "task_query",
        "response",
        "system_prompt",
        "tool_schema",
        "tool_call_summary",
        "collaboration",
        "workspace",
        "artifact",
        "completion_status"
      ],
      "tags": [
        "WorkBuddy",
        "Task Tree",
        "Multi-agent",
        "Workspace"
      ],
      "viewer_url": "/mixed-trajectories/viewer.html",
      "schema_url": "/data/schemas/mixed-collaborative-trajectories.schema.json",
      "quality": {
        "metric": "completion_status",
        "label": "状态边界",
        "label_en": "Completion boundary",
        "status_label": "运行状态",
        "status_label_en": "Run status",
        "score": 8,
        "scale": 12,
        "display": "可追踪",
        "display_en": "Traceable",
        "basis": "8 条 done；4 条 partial 均记录超时边界",
        "basis_en": "8 done trajectories; all 4 partial trajectories retain their timeout boundaries"
      },
      "preview": {
        "label": "FRAMEWORK",
        "summary": "WorkBuddy",
        "sequence": [
          "query",
          "delegate",
          "tool",
          "merge",
          "workspace"
        ],
        "metrics": [
          {
            "label": "任务编排",
            "label_en": "Task orchestration",
            "value": "Task Tree"
          },
          {
            "label": "协作结构",
            "label_en": "Collaboration structure",
            "value": "主 / 子 Agent",
            "value_en": "Primary / subagent"
          },
          {
            "label": "交付形态",
            "label_en": "Delivery form",
            "value": "Workspace"
          }
        ],
        "compare": [
          {
            "label": "轨迹状态",
            "label_en": "Trajectory status",
            "left": "4 partial",
            "right": "8 done",
            "left_pct": 33,
            "right_pct": 67
          }
        ],
        "note": "10/12 条捕获完整 System Prompt；7/8 条协作任务达到最低子代理要求。",
        "note_en": "10/12 trajectories retain the complete System Prompt. 7/8 collaborative tasks meet the minimum subagent requirement."
      },
      "content": {
        "label": "框架结构",
        "label_en": "Framework structure",
        "excerpt": "WorkBuddy 以任务树组织多轮任务，在指定节点声明协作策略与角色分工，并通过 CodeBuddy Code 运行时记录工具过程、结果合并和最终工作区。",
        "excerpt_en": "WorkBuddy organizes multi-round tasks as a task tree, declares collaboration policies and role assignments at selected nodes, and uses the CodeBuddy Code runtime to record tool execution, result merges, and the final workspace.",
        "inventory": [
          "WorkBuddy 任务树驱动的多轮任务演进",
          "按任务节点配置角色化协作与结果合并",
          "CodeBuddy Code 运行时的工具调用与状态记录",
          "最终工作区、交付文件和 done / partial 边界"
        ],
        "inventory_en": [
          "Multi-round task progression orchestrated by the WorkBuddy task tree",
          "Role-based collaboration and result merging configured per task node",
          "Tool calls and runtime state captured from the CodeBuddy Code backend",
          "Final workspaces, deliverables, and explicit done / partial boundaries"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "done and partial status",
          "system prompt capture",
          "tool schema coverage",
          "collaboration minimum",
          "round completion",
          "workspace deliverables"
        ]
      },
      "record_count": 12,
      "record_unit": "trajectories",
      "sample_count": 12,
      "product_line": "high_quality_trajectory",
      "download": {
        "filename": "greatruth-mixed-collaborative-trajectories-12.zip",
        "media_type": "application/zip",
        "url": "/downloads/greatruth-mixed-collaborative-trajectories-12.zip",
        "content_url": "https://display.greatruth.cloud/downloads/greatruth-mixed-collaborative-trajectories-12.zip",
        "byte_size": 124810232,
        "sha256": "0ccd60c76883bf736ac5eb30397ecc83b36bad77bea6018c775ec9d40731e628",
        "encoding": "binary",
        "requires_authentication": false
      },
      "step_count": 15865,
      "observability": {
        "source_event_count": 15865,
        "normalized_event_count": 78,
        "tool_call_count": 7287,
        "tool_result_count": 8183,
        "runtime_error_count": 409,
        "subagent_call_count": 45,
        "workspace_file_count": 2671,
        "system_prompt_count": 10
      },
      "artifacts": [
        {
          "role": "curated_sample",
          "title": "Mixed collaborative trajectory display sample",
          "filename": "mixed-trajectories-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/mixed-collaborative-trajectories.schema.json",
          "url": "/data/synthetic/mixed-trajectories-sample.json",
          "content_url": "https://display.greatruth.cloud/data/synthetic/mixed-trajectories-sample.json",
          "byte_size": 1452200,
          "sha256": "57f97758ee6f7ab008eb0af5eab06d9850968973c8b51042fc7a9a1b74bee6ee",
          "record_count": 12,
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ]
    },
    {
      "id": "gr-vertical-coding-cl-t8-003-lean",
      "slug": "vertical-coding-cl-t8-003-lean",
      "title": "无限布尔序列的 Cantor 对角线证明",
      "title_en": "Cantor diagonal proof for infinite Boolean sequences",
      "eyebrow": "VERTICAL · CODING",
      "eyebrow_en": "VERTICAL · CODING",
      "description": "MATH-031 直接选取 Cantor 对角线论证，不添加业务系统或线上事件背景。输入的二维布尔表把每个自然数 row 对应到一条无限布尔序列，Enumerates 声称这些行覆盖全部序列；标准交付物是一个对任意 table 成立的不可枚举性 theorem。证明需要构造对角序列、取得假设中的行号见证、在该行号处读取函数等式，并对布尔值分类得到矛盾。这个片段很短，但每一步都有清楚的数学对应关系，适合作为 Lean4 数理逻辑证明的独立交付。",
      "description_en": "MATH-031 uses the Cantor diagonal argument directly, without adding a business system or production incident. Each natural-number row of a two-dimensional Boolean table represents an infinite Boolean sequence, while Enumerates claims that the rows cover every sequence. The deliverable is a non-enumerability theorem valid for any table. The proof constructs a diagonal sequence, obtains a row witness from the assumption, applies the function equality at that row, and derives a contradiction by considering both Boolean values. Each step has a clear mathematical interpretation, making the task a self-contained Lean4 logic proof.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "Coding 交叉长尾",
        "scope_en": "Coding 交叉长尾"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "coding"
      ],
      "viewer_url": "/vertical-coding-cl-t8-003-lean/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "56"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "13"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "MATH-031 直接选取 Cantor 对角线论证，不添加业务系统或线上事件背景。输入的二维布尔表把每个自然数 row 对应到一条无限布尔序列，Enumerates 声称这些行覆盖全部序列；标准交付物是一个对任意 table 成立的不可枚举性 theorem。证明需要构造对角序列、取得假设中的行号见证、在该行号处读取函数等式，并对布尔值分类得到矛盾。这个片段很短，但每一步都有清楚的数学对应关系，适合作为 Lean4 数理逻辑证明的独立交付。",
        "excerpt_en": "MATH-031 直接选取 Cantor 对角线论证，不添加业务系统或线上事件背景。输入的二维布尔表把每个自然数 row 对应到一条无限布尔序列，Enumerates 声称这些行覆盖全部序列；标准交付物是一个对任意 table 成立的不可枚举性 theorem。证明需要构造对角序列、取得假设中的行号见证、在该行号处读取函数等式，并对布尔值分类得到矛盾。这个片段很短，但每一步都有清楚的数学对应关系，适合作为 Lean4 数理逻辑证明的独立交付。",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-coding-cl-t8-003-lean-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-coding-cl-t8-003-lean-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-coding-cl-t8-003-lean-sample.json",
          "byte_size": 65011,
          "sha256": "491559a3955ed648403abe50ba270d65b819d0d74fef18b3b10a6da1a7df370e",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-coding-cl-t8-003-lean-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-coding-cl-t8-003-lean-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-coding-cl-t8-003-lean-guide.pdf",
          "byte_size": 724102,
          "sha256": "401697ce5e4845afda9fe1ffa635b3792c2ba60e24537a5d5baeabe76e09e6ab",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-coding-cl-t8-003-lean.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-coding-cl-t8-003-lean.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-coding-cl-t8-003-lean.zip",
        "byte_size": 46403,
        "sha256": "d386597f16f2937dea9e0ae19e838b2d8e8fb63a6833c31ca8bef3bf67a795ca",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-coding-vhdl-t8-001-fifo",
      "slug": "vertical-coding-vhdl-t8-001-fifo",
      "title": "工业视觉采集板双时钟可提交包 FIFO",
      "title_en": "Industrial vision acquisition board dual-clock packet-commit FIFO",
      "eyebrow": "VERTICAL · CODING",
      "eyebrow_en": "VERTICAL · CODING",
      "description": "产线 A-17 的 SN-CAM-042 采集链路横跨 96 MHz 写域和 148.5 MHz DMA 读域，现场另有 2051 字异常行要在 last 后重新对齐。候选人从约 130 行的标准异步 FIFO 骨架开始，补齐包提交计数、撤销点、输出预取、容量状态和超长包丢弃；两个公开 testbench 加四组验收侧时序场景覆盖 4 字与 8 字深度。",
      "description_en": "The SN-CAM-042 acquisition link on production line A-17 crosses a 96 MHz write domain and a 148.5 MHz DMA read domain. An abnormal 2,051-word line must be realigned after last. Starting from an approximately 130-line asynchronous FIFO skeleton, the candidate implements packet-commit counting, rollback points, output prefetch, capacity status, and oversized-packet dropping. Two public testbenches and four acceptance-side timing scenarios cover depths of four and eight words.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "Coding 交叉长尾",
        "scope_en": "Coding 交叉长尾"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "coding"
      ],
      "viewer_url": "/vertical-coding-vhdl-t8-001-fifo/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "40"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "16"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "产线 A-17 的 SN-CAM-042 采集链路横跨 96 MHz 写域和 148.5 MHz DMA 读域，现场另有 2051 字异常行要在 last 后重新对齐。候选人从约 130 行的标准异步 FIFO 骨架开始，补齐包提交计数、撤销点、输出预取、容量状态和超长包丢弃；两个公开 testbench 加四组验收侧时序场景覆盖 4 字与 8 字深度。",
        "excerpt_en": "产线 A-17 的 SN-CAM-042 采集链路横跨 96 MHz 写域和 148.5 MHz DMA 读域，现场另有 2051 字异常行要在 last 后重新对齐。候选人从约 130 行的标准异步 FIFO 骨架开始，补齐包提交计数、撤销点、输出预取、容量状态和超长包丢弃；两个公开 testbench 加四组验收侧时序场景覆盖 4 字与 8 字深度。",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-coding-vhdl-t8-001-fifo-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-coding-vhdl-t8-001-fifo-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-coding-vhdl-t8-001-fifo-sample.json",
          "byte_size": 99422,
          "sha256": "c0a693c9afb8df8b2c13cf7c1f6115d0636aeb5b440836b9323f0a2a5266e250",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-coding-vhdl-t8-001-fifo-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-coding-vhdl-t8-001-fifo-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-coding-vhdl-t8-001-fifo-guide.pdf",
          "byte_size": 695511,
          "sha256": "1db789901da3f1bfde8dd4da33d1a0025693bd98f49776fab061d14350450a12",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-coding-vhdl-t8-001-fifo.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-coding-vhdl-t8-001-fifo.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-coding-vhdl-t8-001-fifo.zip",
        "byte_size": 160882,
        "sha256": "68339ef8c0add3d4a0aea422c8f108f0dde9715fd161a33c0fb2ee52cc159663",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-pm-t8-004",
      "slug": "vertical-pm-t8-004",
      "title": "发布组合决策任务",
      "title_en": "Release portfolio decision task",
      "eyebrow": "VERTICAL · 产品经理",
      "eyebrow_en": "VERTICAL · 产品经理",
      "description": "场景模拟匿名维护团队在固定两周发布窗口内完成一次可审计的版本组合决策。任务要求从包内全量冻结需求与讨论证据中发现 16 个工作流映射并处理高风险近邻混淆，调和重复或冲突的指标观测，在团队容量、共享 QA、安全评审、风险预算、硬依赖、互斥方案和 binding commitment 约束下比较三套基线组合。每套基线及八个事件状态不仅要给出最优集合，还要证明 winner/runner-up 的 B/U/N/R、第一 lexicographic separator、regret 和资源 slack；四个反事实还需区分题设变化与最小充分阈值。真实性由完整可追溯的包内证据链、可复算的确定性规划参数，以及与真实发布治理一致的决策、升级、验证、回滚和责任交接共同保障。十项交付物可直接支持 go/no-go、范围重排和条件变化后的复核；包内规划参数仅作为题设控制，不被表述为当前事实。",
      "description_en": "An anonymized maintenance team must make an auditable release-portfolio decision within a fixed two-week window. The task requires identifying 16 workflow mappings from frozen requirements and discussion evidence, resolving high-risk near-neighbor confusion and conflicting metric observations, and comparing three baseline portfolios under capacity, shared QA, security review, risk, dependency, exclusivity, and binding-commitment constraints. For each baseline and eight event states, the deliverables establish the optimal set, winner and runner-up B/U/N/R values, first lexicographic separator, regret, and resource slack. Four counterfactuals distinguish changes to the assumptions from minimum sufficient thresholds. Ten deliverables support go/no-go decisions, scope changes, and later review through traceable evidence, reproducible planning parameters, escalation, validation, rollback, and ownership handoff. Planning parameters are scenario controls rather than claims about current conditions.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "产品经理/互联网运营",
        "scope_en": "产品经理/互联网运营"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "产品经理"
      ],
      "viewer_url": "/vertical-pm-t8-004/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "224"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "43"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "场景模拟匿名维护团队在固定两周发布窗口内完成一次可审计的版本组合决策。任务要求从包内全量冻结需求与讨论证据中发现 16 个工作流映射并处理高风险近邻混淆，调和重复或冲突的指标观测，在团队容量、共享 QA、安全评审、风险预算、硬依赖、互斥方案和 binding commitment 约束下比较三套基线组合。每套基线及八个事件状态不仅要给出最优集合，还要证明 winner/runner-up 的 B/U/N/R、第一 lexicographic separator、regret 和资源 slack；四个反事实还需区分题设变化与最小充分阈值。真实性由完整可追溯的包内证据链、可复算的确定性规划参数，以及与真实发布治理一致的决策、升级、验证、回滚和责任交接共同保障。十项交付物可直接支持 go/no-go、范围重排和条件变化后的复核；包内规划参数仅作为题设控制，不被表述为当前事实。",
        "excerpt_en": "场景模拟匿名维护团队在固定两周发布窗口内完成一次可审计的版本组合决策。任务要求从包内全量冻结需求与讨论证据中发现 16 个工作流映射并处理高风险近邻混淆，调和重复或冲突的指标观测，在团队容量、共享 QA、安全评审、风险预算、硬依赖、互斥方案和 binding commitment 约束下比较三套基线组合。每套基线及八个事件状态不仅要给出最优集合，还要证明 winner/runner-up 的 B/U/N/R、第一 lexicographic separator、regret 和资源 slack；四个反事实还需区分题设变化与最小充分阈值。真实性由完整可追溯的包内证据链、可复算的确定性规划参数，以及与真实发布治理一致的决策、升级、验证、回滚和责任交接共同保障。十项交付物可直接支持 go/no-go、范围重排和条件变化后的复核；包内规划参数仅作为题设控制，不被表述为当前事实。",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-pm-t8-004-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-pm-t8-004-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-pm-t8-004-sample.json",
          "byte_size": 139008,
          "sha256": "9be4696abe2b5bf4793ae3f5081627feec102bad9c7d3dbecf17aaad1415682d",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-pm-t8-004-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-pm-t8-004-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-pm-t8-004-guide.pdf",
          "byte_size": 755933,
          "sha256": "2f1ede3a36e382ab8691c7241a2565282a7b64c9ede1ce9c877ba5958f4de59a",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-pm-t8-004.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-pm-t8-004.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-pm-t8-004.zip",
        "byte_size": 418185,
        "sha256": "a83afa5c6e702c5221edf92b9ec26d1daf34770d4e6281355bb1050c35395693",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-cons-t4-019",
      "slug": "vertical-cons-t4-019",
      "title": "咨询-CONS-T4-019-经营指标树",
      "title_en": "Consulting-CONS-T4-019-Operating Driver Tree",
      "eyebrow": "VERTICAL · 咨询",
      "eyebrow_en": "VERTICAL · 咨询",
      "description": "围绕虚构客户 Northstar Café Group 的 Q2 经营复盘，分析收入、经营动量与贡献利润差异。样例要求拆解销售与利润驱动，核验冲突证据和数据恢复情景。最终将分析转化为带有依赖关系、责任人和触发阈值的分阶段行动方案。",
      "description_en": "Review Q2 performance for the fictional Northstar Café Group, focusing on revenue, operating momentum, and contribution-profit variance. The sample requires driver decomposition, evidence arbitration, and quantified data-recovery scenarios. Findings must become a phased action plan with dependencies, owners, and trigger thresholds.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "咨询",
        "scope_en": "咨询"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "咨询"
      ],
      "viewer_url": "/vertical-cons-t4-019/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "45"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "0"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "咨询-CONS-T4-019-经营指标树",
        "excerpt_en": "咨询-CONS-T4-019-经营指标树",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-cons-t4-019-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-cons-t4-019-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-cons-t4-019-sample.json",
          "byte_size": 132722,
          "sha256": "274e254dac5d1646c6e6cb7e5e0e796d5a41cb0046e879d21cd618054fb0aefb",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-cons-t4-019-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-cons-t4-019-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-cons-t4-019-guide.pdf",
          "byte_size": 904249,
          "sha256": "2f2a305be2dde0905f4f09e3498d7b5830554f60bfa73bd675529ef7417dfaef",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-cons-t4-019.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-cons-t4-019.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-cons-t4-019.zip",
        "byte_size": 122489,
        "sha256": "09068f66b8bd4bc9f248ebef0af40aa68a7c8e60ef115b35c88f215abc58fe8e",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-hv3-cons-0002",
      "slug": "vertical-hv3-cons-0002",
      "title": "咨询-HV3-CONS-0002-史基浦双重重要性",
      "title_en": "Consulting-HV3-CONS-0002-Schiphol Double Materiality",
      "eyebrow": "VERTICAL · 咨询",
      "eyebrow_en": "VERTICAL · 咨询",
      "description": "以 Royal Schiphol Group 为对象，基于年度报告、运营序列、政策文件、研究与利益相关方记录开展双重重要性重评。样例覆盖影响、风险与机遇的证据核验、评分控制和待决事项。最终输出面向 2027 年经营计划、目标体系与披露范围的决策建议。",
      "description_en": "Reassess double materiality for Royal Schiphol Group using annual reports, operating series, policies, studies, and stakeholder records. The sample covers evidence validation for impacts, risks, and opportunities, including scoring controls and open challenges. The deliverable recommends decisions for the 2027 operating plan, target system, and disclosure scope.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "咨询",
        "scope_en": "咨询"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "咨询"
      ],
      "viewer_url": "/vertical-hv3-cons-0002/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "71"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "0"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "咨询-HV3-CONS-0002-史基浦双重重要性",
        "excerpt_en": "咨询-HV3-CONS-0002-史基浦双重重要性",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-hv3-cons-0002-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-hv3-cons-0002-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-hv3-cons-0002-sample.json",
          "byte_size": 214675,
          "sha256": "95bd066d2ee9c4fe87041930186a28971bb0c1d49a18ec47064b7c84349aa03c",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-hv3-cons-0002-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-hv3-cons-0002-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-hv3-cons-0002-guide.pdf",
          "byte_size": 655237,
          "sha256": "3b883c81e1d94c8e824828f269ccfddceea1737e33cc166e36dd768ff0235d2d",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-hv3-cons-0002.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-hv3-cons-0002.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-hv3-cons-0002.zip",
        "byte_size": 140549442,
        "sha256": "3e27402a7ba42198b183540530a35524b2207db807cdb45b0c5b2d7d41ce1ea9",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-log-t8-l4",
      "slug": "vertical-log-t8-l4",
      "title": "末端配送三次关账、证据审计与压力情景应急排程",
      "title_en": "Last-mile delivery: three closes, evidence audit, and stress-scenario recovery scheduling",
      "eyebrow": "VERTICAL · 物流",
      "eyebrow_en": "VERTICAL · 物流",
      "description": "该题映射区域物流控制塔连续关账与应急恢复。全量路线材料用于诊断，冻结运营控制用于三次执行关账；W3 的28份独立案卷包含成本、收益和抵扣证据链，W4在天气、承运能力、数据中断和劳动力短缺压力下进行备援司机与干预方案选择。交付物覆盖路线与站点审计、执行证据、稳健排程、资金与碳控制、收益桥和终态登记。",
      "description_en": "The task models consecutive operational closes and recovery planning in a regional logistics control tower. Complete route materials support diagnosis, while frozen operating controls govern three execution closes. W3 contains 28 independent case files with cost, benefit, and credit evidence chains; W4 requires selecting backup drivers and interventions under weather, carrier-capacity, data-outage, and labor-shortage scenarios. Deliverables cover route and station audits, execution evidence, robust scheduling, financial and carbon controls, benefit reconciliation, and final-state records.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "物流",
        "scope_en": "物流"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "物流"
      ],
      "viewer_url": "/vertical-log-t8-l4/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "168"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "40"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "该题映射区域物流控制塔连续关账与应急恢复。全量路线材料用于诊断，冻结运营控制用于三次执行关账；W3 的28份独立案卷包含成本、收益和抵扣证据链，W4在天气、承运能力、数据中断和劳动力短缺压力下进行备援司机与干预方案选择。交付物覆盖路线与站点审计、执行证据、稳健排程、资金与碳控制、收益桥和终态登记。",
        "excerpt_en": "该题映射区域物流控制塔连续关账与应急恢复。全量路线材料用于诊断，冻结运营控制用于三次执行关账；W3 的28份独立案卷包含成本、收益和抵扣证据链，W4在天气、承运能力、数据中断和劳动力短缺压力下进行备援司机与干预方案选择。交付物覆盖路线与站点审计、执行证据、稳健排程、资金与碳控制、收益桥和终态登记。",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-log-t8-l4-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-log-t8-l4-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-log-t8-l4-sample.json",
          "byte_size": 117400,
          "sha256": "f755e6b2225c979b1e34277247a9c7870aca9ce283b3d58a7f2ee3f7106a234b",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-log-t8-l4-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-log-t8-l4-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-log-t8-l4-guide.pdf",
          "byte_size": 779090,
          "sha256": "ddd7e5a3148fa04c76620ffb51fe296de8664bc371c67d2e32ad622fe1c53b8e",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-log-t8-l4.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-log-t8-l4.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-log-t8-l4.zip",
        "byte_size": 33668212,
        "sha256": "893f64a015e3ab2adb21ff83000eaf35369b545f4b9f744a263bce6c50261d9b",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-ecom-t1-001-gmv",
      "slug": "vertical-ecom-t1-001-gmv",
      "title": "电商-Ecom-T1-001-跨市场GMV",
      "title_en": "E-commerce-Ecom-T1-001-Cross-Market GMV",
      "eyebrow": "VERTICAL · 电商",
      "eyebrow_en": "VERTICAL · 电商",
      "description": "针对 UK、DACH、WEST、NORTH 四个市场的 32 个市场-SKU 单元，完成同比 GMV 异动诊断与全量来源覆盖审计。样例要求对订单、漏斗、广告、库存、促销和退款进行归因对账，并登记每个单元的恢复主因。随后按 CP1 至 CP4 顺序处理执行反馈、调拨、结算与资源再排程。",
      "description_en": "Diagnose year-over-year GMV anomalies across 32 market-SKU units in UK, DACH, WEST, and NORTH, with a complete source-coverage audit. Reconcile order, funnel, advertising, inventory, promotion, and refund evidence, and register a recovery cause for every unit. Then process execution feedback, transfers, settlements, and resource replanning through CP1 to CP4.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "电商",
        "scope_en": "电商"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "电商"
      ],
      "viewer_url": "/vertical-ecom-t1-001-gmv/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "108"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "0"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "电商-Ecom-T1-001-跨市场GMV",
        "excerpt_en": "电商-Ecom-T1-001-跨市场GMV",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-ecom-t1-001-gmv-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-ecom-t1-001-gmv-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-ecom-t1-001-gmv-sample.json",
          "byte_size": 217168,
          "sha256": "0497b5c66eadc53136d1085750e2dd9b3e3f402d1f49b67582184158d7d052a8",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-ecom-t1-001-gmv-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-ecom-t1-001-gmv-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-ecom-t1-001-gmv-guide.pdf",
          "byte_size": 880528,
          "sha256": "d86937ff8fb36c1ce9f5e8bd07c65c628712a87d7146284df7a7196fd56b62c2",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-ecom-t1-001-gmv.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-ecom-t1-001-gmv.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-ecom-t1-001-gmv.zip",
        "byte_size": 63081483,
        "sha256": "34395130f0853cf50c9b1e7943ded9cf3450459fa49dea9d7b51abcd72634d6b",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-cy-t8-001",
      "slug": "vertical-cy-t8-001",
      "title": "企业 Windows 事件日志的事件级攻击链调查与应急响应",
      "title_en": "Event-level attack-chain investigation and incident response in enterprise Windows logs",
      "eyebrow": "VERTICAL · 网安",
      "eyebrow_en": "VERTICAL · 网安",
      "description": "本题模拟企业 SOC 在未知攻击时间窗、受影响主机和 IOC 的条件下，对 155350 条 Windows、Sysmon、Security 与 PowerShell 靶场日志开展完整威胁狩猎。分析员需要先建立日志画像和候选活动簇，再通过 EventUid、ProcessGuid、账号、注册表对象、网络目标和时间顺序验证攻击链，最终交付事件清单、攻击图和应急响应报告。任务具有多阶段状态继承、跨产物一致性和较高证据约束，属于 L4 强长程任务。",
      "description_en": "An enterprise SOC conducts a complete threat hunt across 155,350 Windows, Sysmon, Security, and PowerShell range-log records without a known attack window, affected-host list, or IOC set. The analyst first profiles the logs and identifies candidate activity clusters, then validates the attack chain through EventUid, ProcessGuid, accounts, registry objects, network destinations, and chronology. Deliverables include an event inventory, an attack graph, and an incident-response report. The task requires state to carry across stages, consistency across artifacts, and strong evidence support, making it an L4 long-horizon task.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "网络安全",
        "scope_en": "网络安全"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "网安"
      ],
      "viewer_url": "/vertical-cy-t8-001/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "115"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "23"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "本题模拟企业 SOC 在未知攻击时间窗、受影响主机和 IOC 的条件下，对 155350 条 Windows、Sysmon、Security 与 PowerShell 靶场日志开展完整威胁狩猎。分析员需要先建立日志画像和候选活动簇，再通过 EventUid、ProcessGuid、账号、注册表对象、网络目标和时间顺序验证攻击链，最终交付事件清单、攻击图和应急响应报告。任务具有多阶段状态继承、跨产物一致性和较高证据约束，属于 L4 强长程任务。",
        "excerpt_en": "本题模拟企业 SOC 在未知攻击时间窗、受影响主机和 IOC 的条件下，对 155350 条 Windows、Sysmon、Security 与 PowerShell 靶场日志开展完整威胁狩猎。分析员需要先建立日志画像和候选活动簇，再通过 EventUid、ProcessGuid、账号、注册表对象、网络目标和时间顺序验证攻击链，最终交付事件清单、攻击图和应急响应报告。任务具有多阶段状态继承、跨产物一致性和较高证据约束，属于 L4 强长程任务。",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-cy-t8-001-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-cy-t8-001-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-cy-t8-001-sample.json",
          "byte_size": 202899,
          "sha256": "cf5e49c3fd286e601398fc67e282ef2d76a7d6250ff6b172e72c6300a56f474f",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-cy-t8-001-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-cy-t8-001-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-cy-t8-001-guide.pdf",
          "byte_size": 797835,
          "sha256": "5e5365a5efac65ad329dcf224168fd4ce663d73a32bf385edd28052f5585ed36",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-cy-t8-001.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-cy-t8-001.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-cy-t8-001.zip",
        "byte_size": 25794090,
        "sha256": "6256af67cff42aff15b1bb3cb2cce10b8e05ccf629b701f88c8bd4250c38a181",
        "encoding": "binary",
        "requires_authentication": false
      }
    },
    {
      "id": "gr-vertical-hv3-gene-0001",
      "slug": "vertical-hv3-gene-0001",
      "title": "通用办公-HV3-GENE-0001-接口续约桥",
      "title_en": "General Office-HV3-GENE-0001-Interface Bridge Renewal",
      "eyebrow": "VERTICAL · 通用办公",
      "eyebrow_en": "VERTICAL · 通用办公",
      "description": "评估 Northstar Regional Health Network（虚构）申请的 90 天接口续约桥接方案，决策截止时间为 2024 年 6 月 21 日。样例要求核对合同期限、服务组件、连续性影响、证据请求、审批与流动性记录。最终需给出批准、附条件批准或拒绝，并明确监控指标、升级路径和回退条件。",
      "description_en": "Evaluate a 90-day interface bridge requested by the fictional Northstar Regional Health Network, using a decision cutoff of June 21, 2024. Verify contract dates, service components, continuity impacts, evidence requests, approvals, and liquidity records. Conclude with approve, approve with conditions, or decline, including monitoring metrics, escalation paths, and rollback conditions.",
      "buyer_type": "垂域任务数据",
      "buyer_type_en": "Vertical task data",
      "benchmark_anchor": {
        "label": "垂域任务",
        "label_en": "Domain task",
        "scope": "通用办公",
        "scope_en": "通用办公"
      },
      "category": "vertical_domain",
      "product_line": "vertical_domain",
      "visual_kind": "env_matrix",
      "version": "2026.09",
      "release_date": "2026-09-14",
      "status": "verified",
      "languages": [
        "zh-CN",
        "en"
      ],
      "modalities": [
        "instruction",
        "environment",
        "deliverables",
        "rubric",
        "evaluation"
      ],
      "tags": [
        "Vertical",
        "通用办公"
      ],
      "viewer_url": "/vertical-hv3-gene-0001/viewer.html",
      "schema_url": "/data/schemas/harbor-env.schema.json",
      "quality": {
        "metric": "source_evaluation",
        "label": "来源评测",
        "label_en": "Source evaluation",
        "display": "按样例口径",
        "display_en": "Per-sample scope",
        "basis": "Source package evaluation materials",
        "status_label": "垂域样例",
        "status_label_en": "Vertical sample"
      },
      "preview": {
        "label": "VERTICAL TASK",
        "label_en": "VERTICAL TASK",
        "summary": "题面与交付",
        "summary_en": "Instruction and deliverables",
        "sequence": [
          "task",
          "input",
          "deliverable",
          "verify"
        ],
        "metrics": [
          {
            "label": "样例",
            "label_en": "Sample",
            "value": "1"
          },
          {
            "label": "归档文件",
            "label_en": "Archive files",
            "value": "99"
          },
          {
            "label": "Rubric",
            "label_en": "Rubric items",
            "value": "0"
          }
        ],
        "note": "完整题面、交付物、评测材料和说明书。",
        "note_en": "Complete instruction, deliverables, evaluation material, and guide."
      },
      "content": {
        "label": "任务说明",
        "label_en": "Task specification",
        "excerpt": "通用办公-HV3-GENE-0001-接口续约桥",
        "excerpt_en": "通用办公-HV3-GENE-0001-接口续约桥",
        "inventory": [
          "完整任务题面",
          "输入材料与环境",
          "参考交付物、Rubric 与评测记录",
          "配套说明书"
        ],
        "inventory_en": [
          "Complete task instruction",
          "Inputs and environment",
          "Reference deliverables, rubric, and evaluation records",
          "Companion guide"
        ]
      },
      "evaluation": {
        "recommended_checks": [
          "instruction completeness",
          "deliverable coverage",
          "rubric and evaluation scope"
        ]
      },
      "record_count": 1,
      "record_unit": "tasks",
      "sample_count": 1,
      "artifacts": [
        {
          "role": "harbor_sample",
          "title": "Structured vertical sample",
          "filename": "vertical-hv3-gene-0001-sample.json",
          "media_type": "application/json",
          "schema_url": "/data/schemas/harbor-env.schema.json",
          "url": "/data/vertical/vertical-hv3-gene-0001-sample.json",
          "content_url": "https://display.greatruth.cloud/data/vertical/vertical-hv3-gene-0001-sample.json",
          "byte_size": 123185,
          "sha256": "46286d48f6b8ecc2e410f1a4520903c178dedd3ac6d3ca4fa9cf16ba0ec82819",
          "encoding": "utf-8",
          "requires_authentication": false
        },
        {
          "role": "guide",
          "title": "Companion guide",
          "filename": "vertical-hv3-gene-0001-guide.pdf",
          "media_type": "application/pdf",
          "url": "/downloads/vertical-hv3-gene-0001-guide.pdf",
          "content_url": "https://display.greatruth.cloud/downloads/vertical-hv3-gene-0001-guide.pdf",
          "byte_size": 670371,
          "sha256": "926a8dba2310fbb4744d7a17c6351ec496b8109672565ffa40df3b22faa856c6",
          "encoding": "utf-8",
          "requires_authentication": false
        }
      ],
      "download": {
        "filename": "vertical-hv3-gene-0001.zip",
        "media_type": "application/zip",
        "url": "/downloads/vertical-hv3-gene-0001.zip",
        "content_url": "https://display.greatruth.cloud/downloads/vertical-hv3-gene-0001.zip",
        "byte_size": 140749,
        "sha256": "7e4b09b85c3d65e7a3e3fdc25954f900faa7cb24238f91e50bc9dc274074e762",
        "encoding": "binary",
        "requires_authentication": false
      }
    }
  ]
}
