{
  "version": "1.1",
  "role": "First dedicated cross-modal Data Ops owner supporting audio, video, world-model and multimodal research; music and audio are the candidate's strongest entry domain.",
  "groups": [
    {
      "id": "source_to_training_data",
      "title": "01. 从数据源到可训练数据",
      "competencies": ["ownership", "data quality", "scale", "cross-functional execution"],
      "minimum_evidence": "一条真实数据源的完整时间线、职责边界、至少一个失败点、一个可安全披露的质量信号。",
      "questions": [
        {"question":"Walk me through one real music-data source from discovery to accepted training data.","guide_zh":"不要概括整个体系。选一个真实来源，按发现、评估、商务/合规、下载、解析、验收、入库的顺序回忆。"},
        {"question":"What problem made this source worth pursuing, and what would have happened if you had done nothing?","guide_zh":"说明业务或模型缺口、当时的约束和为什么值得投入。"},
        {"question":"Which decisions were personally yours, and which belonged to procurement, legal, infrastructure, suppliers, or model researchers?","guide_zh":"画清职责边界，避免把团队成果全部说成个人成果。"},
        {"question":"What failed first, and how did you discover it?","guide_zh":"必须是真实故障或明确回答暂时想不起；不要用假设故事代替。"},
        {"question":"What trade-off did you make between quality, coverage, cost, and speed?","guide_zh":"回忆一个你必须舍弃某些东西的实际决策。"},
        {"question":"What changed because of your intervention, and what evidence can you safely share?","guide_zh":"结果可用规模区间、通过率变化、下游采用或评测变化表达；不泄露供应商和合同信息。"},
        {"question":"What would you do differently now, and which parts of this sourcing workflow would need fresh checks in another modality?","guide_zh":"提炼反思，再具体说明新模态的来源、质量和专家接口需要怎样重查；不要默认跨模态岗位的首个项目就是音视频联合数据。"}
      ]
    },
    {
      "id": "acceptance_pipeline",
      "title": "02. 从零搭建 SFT 验收流水线",
      "competencies": ["pipeline design", "precision-recall", "quality operations", "research partnership"],
      "minimum_evidence": "10万级 SFT 背景、分层检测逻辑、一个阈值权衡、人工复核边界和下游采用证据。",
      "questions": [
        {"question":"Tell me why the SFT acceptance pipeline had to be built from zero.","guide_zh":"之前哪里不可用，风险是什么，为什么不能只靠人工抽听。"},
        {"question":"Describe the pipeline stages and the purpose of each stage.","guide_zh":"按去重、元数据-音频匹配、伪无损、内容/质量检查、人工复核等真实顺序回答。"},
        {"question":"Which checks were hard gates, which were scores, and which required human review?","guide_zh":"给出分层逻辑，避免声称所有判断都能自动化。"},
        {"question":"How did you tune false acceptance versus false rejection?","guide_zh":"回忆 gold sample、抽样、下游风险和阈值调整；不知道具体数字可以说方法与边界。"},
        {"question":"Describe one edge case that broke the original rule set.","guide_zh":"例如版本、重编码、错误元数据或伪无损的真实边界案例。"},
        {"question":"How did you work with model researchers to decide whether better acceptance actually improved training data?","guide_zh":"说明双方交付物、决策节奏和如何避免只优化离线通过率。"},
        {"question":"How would you extend this pipeline to synchronized video, dialogue, effects, ambience, and music?","guide_zh":"拆分音轨存在性、同步、声源绑定、内容一致性、版权与技术质量。"}
      ]
    },
    {
      "id": "structured_captioning",
      "title": "03. 十维分段结构化描述",
      "competencies": ["taxonomy", "annotation design", "audio-language alignment", "model impact"],
      "minimum_evidence": "一个旧标签失败例子、十维设计逻辑、一个歧义规则、人工与模型分工、评测边界。",
      "questions": [
        {"question":"What was wrong with the previous music tags? Give one concrete before-and-after example.","guide_zh":"选一个真实样本，说明旧标签遗漏了什么训练信号。"},
        {"question":"How did you choose the ten dimensions and section-level representation?","guide_zh":"说明设计原则，不必背出不能公开的内部字段。"},
        {"question":"Which dimensions were hardest for annotators to agree on, and why?","guide_zh":"回忆主观性、时间边界、多层结构或音乐术语歧义。"},
        {"question":"How did you handle overlapping sections, transitions, or attributes that changed gradually?","guide_zh":"给出一条真实标注规则或明确哪些细节不能披露。"},
        {"question":"What did the model pre-annotate, and what did humans verify or correct?","guide_zh":"讲清人机协作，不要把 captioner 能力归到自己训练模型。"},
        {"question":"How did you know the new captions were more useful than plausible-looking text?","guide_zh":"区分文本自然度、音频 groundedness、主观评测和下游训练价值。"},
        {"question":"How would this experience help define temporally grounded audio descriptions for generated video?","guide_zh":"连接事件、声源、时间、空间、叙事作用和视频上下文。"}
      ]
    },
    {
      "id": "evaluation_release_gate",
      "title": "04. 发布门禁与竞品横评",
      "competencies": ["evaluation science", "statistics", "quality judgment", "decision making"],
      "minimum_evidence": "10+主观集、常规 CMOS 与竞品 MOS 的差异、一个评分 QA 机制、一次真实发布决策。",
      "questions": [
        {"question":"How did you structure the evaluation suite that gated model releases?","guide_zh":"按能力、语言、风格、使用场景和主客观轨道解释。"},
        {"question":"Why did routine tests use three-rater CMOS while competitor comparisons used ten-rater MOS?","guide_zh":"说明灵敏度、绝对质量、预算、方差与使用目的。"},
        {"question":"How did you control loudness, order, rater fatigue, genre preference, and low-quality raters?","guide_zh":"选最熟悉的两三个机制展开，不要只罗列名词。"},
        {"question":"Tell me about one release decision that changed because of evaluation evidence.","guide_zh":"给出候选版本、风险、证据、你提出的建议和最终结果。"},
        {"question":"How did you treat disagreement and uncertainty instead of hiding them in an average?","guide_zh":"说明分 slice、置信区间、复评或专家仲裁的真实做法。"},
        {"question":"State the Suno and SongEval claims precisely, including what you are not claiming.","guide_zh":"明确 SongEval 是公开 benchmark，你没有创建它；区分模型团队成果与个人贡献。"},
        {"question":"Design a release gate for a model that jointly generates video, dialogue, effects, ambience, and music.","guide_zh":"既要分模态评估，也要评音画关系、可控性和生产稳定性。"}
      ]
    },
    {
      "id": "reward_preference_data",
      "title": "05. 偏好数据与近 90% 人类一致率",
      "competencies": ["preference data", "rubric calibration", "team leadership", "iteration"],
      "minimum_evidence": "偏好任务单位、冲突维度处理、一致率定义或边界、困难样本迭代和个人职责。",
      "questions": [
        {"question":"What exactly did one reward-model preference task ask a rater to decide?","guide_zh":"描述一个标注单元、输入、比较对象和输出，不暴露内部名称。"},
        {"question":"How did the rubric handle musicality, prompt following, audio quality, and taste disagreement?","guide_zh":"说明多维冲突时如何判断，不要把审美分歧都当噪声。"},
        {"question":"What did nearly ninety percent human agreement mean operationally?","guide_zh":"如果记不清分母、评估集或计算方法，明确标未知，不能只重复简历数字。"},
        {"question":"How did you distinguish a difficult example from a broken annotation task?","guide_zh":"讲抽样、专家复核、歧义率或 rubric 修订。"},
        {"question":"Which labeling-system change produced the biggest measurable improvement?","guide_zh":"回忆一次真实迭代和前后表现。"},
        {"question":"How did you keep a roughly ten-person operation calibrated over time?","guide_zh":"说明培训、校准、升级、返工和移除低质量人员的机制。"},
        {"question":"What new preference tasks would be needed for audio-video generation?","guide_zh":"考虑声音内容、同步、空间、叙事作用、画面质量与主观偏好的冲突。"}
      ]
    },
    {
      "id": "automatic_judge_bias",
      "title": "06. 自动评测、LLM-as-Judge 与偏差",
      "competencies": ["automatic evaluation", "calibration", "bias analysis", "human-in-the-loop"],
      "minimum_evidence": "80%+适用维度、gold 构造、一个 prompt 设计决策、一个偏差 slice、人工 gold 边界。",
      "questions": [
        {"question":"Which instruction-following dimensions reached more than eighty percent agreement with human gold?","guide_zh":"说清是文本可判断维度，不要扩大到整体音乐审美。"},
        {"question":"How did you build and protect the human-gold calibration set?","guide_zh":"说明样本、切分、复核和避免把调参集当测试集。"},
        {"question":"What judge-prompt decision did you personally make, and why?","guide_zh":"回忆输出结构、证据要求、分解维度、反偏置或失败处理。"},
        {"question":"How did you inspect disagreement by slice rather than only aggregate agreement?","guide_zh":"给出语言、风格、长度、复杂约束或音质等真实 slice。"},
        {"question":"How did you detect pop bias in aesthetic or musicality scorers?","guide_zh":"说明什么反例让你意识到问题，以及为何只作为 guardrail。"},
        {"question":"Tell me about a case where the automatic score was confident and wrong.","guide_zh":"如果没有可确认案例，明确说需要回忆，不要现场编造。"},
        {"question":"How would you calibrate an audio-video judge against expert human preference?","guide_zh":"区分硬约束、视觉审美、音频质量、音画关系和 pairwise preference。"}
      ]
    },
    {
      "id": "composition_recovery",
      "title": "07. 数据配比诊断与能力恢复",
      "competencies": ["causal diagnosis", "data composition", "experiment design", "musical expertise"],
      "minimum_evidence": "Latin/Blues 的真实听感问题、竞争假设、数据干预、目标 slice 恢复和回归检查。",
      "questions": [
        {"question":"What did weak Latin grooves or blues vocals actually sound like?","guide_zh":"用可听见的失败描述，不要只说效果不好。"},
        {"question":"Why did you suspect data composition rather than model capacity, representation, or prompting?","guide_zh":"列出竞争假设和排除依据。"},
        {"question":"What evidence in the data distribution supported your hypothesis?","guide_zh":"说可安全披露的比例、覆盖或质量观察，不泄露精确内部总量。"},
        {"question":"What intervention did you make, and what did you deliberately keep unchanged?","guide_zh":"说明采样或定向数据的变化，以及控制变量。"},
        {"question":"How did you evaluate recovery and guard against regressions elsewhere?","guide_zh":"包含目标 slice、人类听评和宽基准回归。"},
        {"question":"What evidence would have falsified your data-composition explanation?","guide_zh":"展示研究思维，而不是事后归因。"},
        {"question":"How would you diagnose a video model whose generated soundtrack repeatedly misses culturally specific scenes?","guide_zh":"迁移到文化语境、声音事件、音乐风格和视频内容绑定。"}
      ]
    },
    {
      "id": "representation_disagreement",
      "title": "08. 表示方案争议与失败判断",
      "competencies": ["conflict", "technical judgment", "influence", "failure learning"],
      "minimum_evidence": "高信息密度假设、实际质量上限证据、你的具体测量、决策影响和归因边界。",
      "questions": [
        {"question":"Why did the new audio representation look better in theory?","guide_zh":"解释信息密度等理论优势，但不要冒充 codec 训练负责人。"},
        {"question":"What did lower audio-quality ceiling mean in audible and measurable terms?","guide_zh":"给出你实际听到或测到的现象。"},
        {"question":"What did you personally design, measure, or present?","guide_zh":"与训练工程师、研究负责人和决策者的工作严格分开。"},
        {"question":"Who disagreed with the evidence, and how did you handle that disagreement?","guide_zh":"如果没有人际冲突，就说技术观点分歧，不要戏剧化。"},
        {"question":"What alternative explanation did you test before blaming the representation?","guide_zh":"考虑数据、解码器、训练目标、评测偏差和实现问题。"},
        {"question":"What decision changed, and what can you honestly claim about saved time or cost?","guide_zh":"没有获批数字就不要说节省了多少，只说避免继续投入或影响方向。"},
        {"question":"How would this lesson apply to audio representations inside a joint video model?","guide_zh":"说明信息压缩、同步、内容可控性和解码质量之间的权衡。"}
      ]
    },
    {
      "id": "product_feedback_loop",
      "title": "09. 用户反馈到模型优先级",
      "competencies": ["product judgment", "online-offline loop", "causal humility", "prioritization"],
      "minimum_evidence": "一个真实用户信号、偏差控制、进入 benchmark 的路径、50% 增长的准确归因边界。",
      "questions": [
        {"question":"Give one concrete example of a user signal becoming a model or evaluation priority.","guide_zh":"从行为或反馈出发，讲到错误分类、评测 slice 和改进优先级。"},
        {"question":"Why was that signal more meaningful than raw popularity or engagement?","guide_zh":"说明曝光、推荐、头部内容和新奇效应等偏差。"},
        {"question":"What did you personally change in the evaluation or feedback scheme?","guide_zh":"区分产品、算法、研究和你的职责。"},
        {"question":"How did offline evaluation and product evidence disagree?","guide_zh":"如果没有真实冲突，明确暂时想不起，不要虚构。"},
        {"question":"How did you decide whether to create a new benchmark slice or change an existing metric?","guide_zh":"解释何时新增、何时修订、何时保持不动。"},
        {"question":"How should you describe the fifty-percent user-growth result without claiming sole causality?","guide_zh":"使用 contributed/associated/steered，明确增长分析可能由其他团队拥有。"},
        {"question":"What user signals would you investigate with researchers and creative partners when transferring this workflow to another modality?","guide_zh":"说明所选创作任务中哪些行为能反映使用价值、哪些仍受曝光或新奇效应影响；先向研究和创意伙伴确认目标，不预设实际项目或已观察到的结果。"}
      ]
    },
    {
      "id": "leadership_collaboration",
      "title": "10. 团队协作、质量压力与影响力",
      "competencies": ["leadership", "conflict resolution", "deadline judgment", "communication"],
      "minimum_evidence": "真实团队结构、一个跨职能决策、一个质量/进度权衡、你的沟通动作和结果；没有故事必须标未知。",
      "questions": [
        {"question":"Describe the real team and stakeholder structure around one project you led.","guide_zh":"不说私人人名；说明标注、供应商、模型、产品、工程等角色和决策权。"},
        {"question":"Tell me about a real disagreement with a researcher, product partner, vendor, or annotator.","guide_zh":"选择你记得细节的冲突；如果没有，先标未知再继续挖掘。"},
        {"question":"What did the other side believe, and why was their position reasonable?","guide_zh":"避免把对方写成错误角色，展示理解和共同目标。"},
        {"question":"What evidence or process did you introduce to move the decision forward?","guide_zh":"具体到对照、抽样、评测、校准会议或升级机制。"},
        {"question":"Tell me about a deadline that forced you to reduce scope without corrupting the quality signal.","guide_zh":"必须是真实案例；没有就标未知，不能把决策框架当历史。"},
        {"question":"What was the outcome, and what part of it was not caused by you?","guide_zh":"给结果，同时主动收紧个人归因。"},
        {"question":"How would you earn trust with researchers in a modality that is new to you while coordinating work without assumed reporting authority?","guide_zh":"说明先交付什么、如何学习模态差异、准确分配功劳与协调决定；第一位 owner 不自动拥有直属团队或研究优先级最终决定权。"}
      ]
    },
    {
      "id": "agent_assisted_tools",
      "title": "11. Coding 与内部工具：从成果追到验证",
      "competencies": ["hands-on delivery", "coding judgment", "verification", "ownership", "cross-modal transfer"],
      "minimum_evidence": "一个实际工具的用途、一条可脱敏输入到输出路径、本人能解释的代码或检查、已发生或已测试的失败及真实使用依据；缺失细节标未知，拟议方法标假设，原始口述与帮助后补充分开记录。",
      "questions": [
        {"question":"Which internal tool best shows how you turned a recurring data or evaluation problem into a useful result, and what does it produce?","guide_zh":"先选一个真实工具讲用途与输出；已确认有 10+ 内部工具，例子包括评测、sourcing 质检和听评工具。数量不证明单个工具的成熟度，不补全未确认的工具清单。"},
        {"question":"Can you trace one input you can safely describe through that tool to the output someone would inspect?","guide_zh":"保持同一个工具，补充一条本人能确认的输入、处理和输出；内部字段、接口、技术选型没有记录就标未知或保密。虚构演示可以帮助解释，但必须与真实运行分开。"},
        {"question":"Which part of that workflow can you read, explain and modify yourself, and where did Codex contribute to the implementation?","guide_zh":"说明自己定义、阅读、修改、检查和调试的具体部分，以及 Agent 的实现贡献。可解释实际 Python 或 SQL；不把 Agent 生成代码算作全部手写，也不由此推定独立训练或分布式系统能力。"},
        {"question":"What did you personally check on that example before accepting the result, and how could the same check expose a plausible but wrong output?","guide_zh":"追到输入约束、中间结果、可手工核对样例或实际测试；说明为什么检查能发现错误。已有操作方法不等于这一次确实做过，未确认的测试结果留待本人补充。"},
        {"question":"Which failure have you actually observed or tested in this tool, and what evidence shows whether a repair worked?","guide_zh":"可以讲实际发生或明确测试过的失败；若没有可确认案例，直接标未知，再把拟议测试单独标假设。不编造事故、修复经过、日志、回归结果或省时数据。"},
        {"question":"What use or feedback from other people can you verify for this tool, and what remains unknown about adoption or impact?","guide_zh":"区分自己使用、同事演示、真实复用和拟议推广；可讲已确认反馈或说明暂无可披露记录。不补造用户数、团队采用率、节省时间、公司级推广责任或正式汇报权。"},
        {"question":"How would you adapt this workflow for a researcher working in a new modality, and which checks would need that researcher's judgment?","guide_zh":"从刚才同一个工具提出迁移方案，说明可复用部分、新模态输入与验收差异、专家接口和最小交付。以条件方案表述，不把新模态工具、人员或研究优先级说成已获批准。"}
      ]
    },
    {
      "id": "reusable_skills",
      "title": "12. 可复用 Skills：把操作判断变成可检查流程",
      "competencies": ["workflow design", "knowledge transfer", "quality checks", "reproducibility", "cross-modal transfer"],
      "minimum_evidence": "一项真实重复工作、一个 Skill 的输入与输出约定、本人编写或审查的关键步骤、一次可核对使用或测试及复用边界；没有记录的细节标未知，拟议改进标假设，提示内容不自动成为独立事实。",
      "questions": [
        {"question":"Which recurring workflow did one of your reusable Skills make easier to repeat, and what useful output does it produce?","guide_zh":"先讲一个真实工作及产物；已确认约 20 个 Skills，把日常经验转成明确的输入、步骤、检查和输出约定。不要把数量直接讲成质量或效率收益。"},
        {"question":"For one invocation you can verify, what inputs did the Skill require and what sequence did it follow to produce the output?","guide_zh":"继续同一个 Skill，沿一次可确认使用追踪输入、步骤和输出；名称、具体源、工具和参数由本人补充。没有运行记录就标未知，不能把理想流程写成执行历史。"},
        {"question":"Which instruction or check captures your own operating judgment, and why is it useful beyond simply saving a prompt?","guide_zh":"选一条本人确实定义或审查的规则，说明它解决了什么重复判断；不虚构 Skill 原文或内部操作步骤。可解释工作流与单条提示的区别，但应回到具体例子。"},
        {"question":"How did you check a particular output from this Skill, and what would let another person reproduce that check?","guide_zh":"补充可核对的输入版本、输出约定、检查方式或复跑记录；仅介绍通用复现方法时明确这是方法。没有版本管理或成功复跑的真实依据，就不能说已经做到。"},
        {"question":"If an input is missing or a source conflicts with the task, what behavior have you actually tested and what still needs verification?","guide_zh":"问题中的缺失或冲突是条件，不表示发生过事故。区分实际测试结果与拟议暂停、澄清、拒收或人工检查；未知行为留待测试，不默认 Skill 自动正确处理。"},
        {"question":"What can you verify about other people reusing this Skill, and how would you decide when its instructions need an update?","guide_zh":"先讲真实可确认的复用，再把维护方案标清。团队采用率、培训计划、正式推广职责和效率收益没有获准事实；没有单个 Skill 的反馈时保留未知。"},
        {"question":"How would you adapt this Skill for a different research modality while deciding which domain rules must be checked again?","guide_zh":"提出一次有范围的迁移：保留通用步骤，重新确认模态字段、专家判断和验收条件。不要默认音频经验或约 20 个 Skills 全部可以原样迁移。"}
      ]
    },
    {
      "id": "research_bots",
      "title": "13. Research Bots：从一期产物追到来源与使用",
      "competencies": ["research prioritization", "source discipline", "human review", "workflow reliability", "cross-modal transfer"],
      "minimum_evidence": "选择一个真实 Bot，一期可确认产物、一条来源到判断的路径、本人复核动作、已确认失败或明确未知及真实使用边界；未测覆盖、准确率和收益不得补造，假设与独立回答分别记录。",
      "questions": [
        {"question":"Which of your research bots produces a useful artifact for your work, and how do you use that artifact?","guide_zh":"先选一个：全球新音乐与趋势、每周音频论文、重要竞品模型动态。已确认多个 Bot，但没有精确总数；讲产物与用途，不默认覆盖全面或已经提高决策质量。"},
        {"question":"Can you trace one edition or update you can verify from its source material through selection and interpretation to the final output?","guide_zh":"继续同一个 Bot，补充一期本人可确认的材料；内部来源列表、连接器、调度和部署结构未知或保密。没有可披露的实际一期，就标未知，不制造论文、事件、日期或链接。"},
        {"question":"Which claim in that output can you trace to an original source and date, and which part is your interpretation?","guide_zh":"选择一条真实可核对内容，分开来源事实和分析判断。没有原材料时不能假装查验；音乐趋势、论文结果和发布宣传也不能互相替代证据。"},
        {"question":"What did you personally review to decide whether this output was useful, and what do you still not know about important material it may have missed?","guide_zh":"说明本人看过的来源或产物与使用决定。未测召回、覆盖或准确率就保留未知；一期读起来有用不能证明重要内容未遗漏。"},
        {"question":"Which incorrect or stale output have you actually observed or tested, and how did you decide whether downstream use could resume?","guide_zh":"有实际事件或明确测试才讲经过；没有就标未知，再把拟议测试方法单独标假设。可以讨论保留来源、人工复核与失败后检查的方法，不能据此补造某次事故或修复效果。"},
        {"question":"What can you verify about who uses this bot's output and who makes the final research or product decision?","guide_zh":"补充这个 Bot 实际可确认的用户与决定接口；若仅自己使用就如实说。没有获准的用户数、采用率、推送时延或商业收益，不把推送等同于团队采用。"},
        {"question":"How would you adapt this bot to support researchers in another modality, and what source or review criteria would need to change?","guide_zh":"作为未来方案，说明研究问题、来源范围、重要性判断和领域复核如何变化。不要默认已有跨模态 Bot、获准来源或另一团队的采用承诺。"}
      ]
    },
    {
      "id": "notation_conversion",
      "title": "14. 乐谱转换工具：从识别结果到可读可检查产物",
      "competencies": ["product delivery", "music judgment", "model integration", "artifact validation", "cross-modal transfer"],
      "minimum_evidence": "转换工具的实际用途、一条可确认识别结果到谱面路径、本人和已有识别模型的分工、实际产物检查及速度与准确率边界；与早期 lead-sheet 标注项目分开，未知细节和拟议方法分别标记。",
      "questions": [
        {"question":"What does your content-to-score tool let someone do with recognized musical content, and which output best shows its value?","guide_zh":"先讲已经做出的用途与产物；已确认依赖现有内部识别模型，转换用时不到一分钟，并支持五线谱、弹唱和弦谱、功能和声谱、简谱和滚动五线谱播放。不要把速度或格式数量当成准确率。"},
        {"question":"Can you trace one verifiable song or recognized-content example from the existing recognition stage to a rendered score?","guide_zh":"只复述本人能确认的识别到转换到谱面路径；共享中间表示或适配器若实际存在再讲，待本人确认。具体字段、库、接口和格式细节也不能由旧预答推定；虚构演示须明确标注。"},
        {"question":"Which parts of this conversion workflow did you own with Codex, and how does that differ from the earlier lead-sheet annotation project?","guide_zh":"本人负责转换和产品流程、集成现有识别模型及产物检查。早期项目是共同制定标注标准并转成 web DAW 需求；不合并两条项目、不声称训练识别模型或独自手写整个工具。"},
        {"question":"Which checks on timing, harmony, readability and scrolling playback did you actually perform on that example, and what did you observe?","guide_zh":"补充本人确实检查过的节拍、调性、和弦、旋律、结构、可读性或静态谱面与播放的一致性。已有检查方法不代表所有检查都在此例完成；没有观察记录就标未知。"},
        {"question":"Which incorrect output have you actually observed or tested, and how did you distinguish recognition errors from conversion or rendering errors?","guide_zh":"先确认是否真有可描述的错误或测试；没有就标未知，分层排查作为拟议方法。不能编造困难歌曲、修复经过、内部模型缺陷或改善比例。"},
        {"question":"What can you verify about the under-one-minute timing and how people use the outputs, while keeping speed separate from transcription accuracy?","guide_zh":"不到一分钟是已批准的当前流程速度表述；具体计时范围、样本条件和使用反馈仍需本人补充。没有获准的整体准确率、用户数或节省工时，也不能声称任意歌曲都完美转写。"},
        {"question":"How would you apply the same model-output-to-reviewable-artifact approach to another modality, and where would you need a domain expert?","guide_zh":"提出有范围的跨模态方案：把模型输出转成可检查产物，同时重新定义输入、表示、显示和专家检查。没有实际完成的跨模态实现或团队承诺就标假设。"}
      ]
    }
  ]
}
