{
  "version": 1,
  "source": "FACT-DEPTH-QUESTIONNAIRE-2026-08-23.md",
  "groups": [
    {
      "id": "sourcing_acceptance",
      "title": "1. Sourcing and acceptance ownership",
      "minimum_evidence": "One end-to-end sequence, a responsibility boundary, one real failure or explicit unknown, and one externally safe validation signal.",
      "questions": [
        "In real chronological order, how did one data source move from discovery to accepted training data?",
        "Which decisions were personally yours, and which belonged to procurement, legal or compliance, model researchers, infrastructure engineers, suppliers, or annotators?",
        "What was the most common rejection reason at each stage?",
        "Which checks were hard gates, which were scores, and which required human review?",
        "What threshold or sampling decision can you safely describe? If none is documented, say that clearly.",
        "Describe one real failure that changed the pipeline. If you do not have a documented incident, say that instead of giving a hypothetical story."
      ]
    },
    {
      "id": "structured_captioning",
      "title": "2. Structured captioning",
      "minimum_evidence": "One before-and-after annotation example, one ambiguity rule, personal scope, and evaluation design without claiming SongEval causality.",
      "questions": [
        "What was wrong with the previous tags? Give me one real example.",
        "How were the ten dimensions defined, and which dimensions were hardest to annotate consistently?",
        "How were section boundaries represented when sections overlapped or were ambiguous?",
        "What did model pre-annotation do, and what exactly did humans verify or correct?",
        "What automation or tooling did you personally build or specify?",
        "What evaluation supports the captioner comparison, and what limitations make that comparison non-universal?"
      ]
    },
    {
      "id": "lead_sheet",
      "title": "3. Lead-sheet annotation",
      "minimum_evidence": "Safe schema detail, one ambiguity policy, tool ownership, and the exact boundary of the subjective SOTA claim.",
      "questions": [
        "Which musical elements were represented: melody, chords, key, meter, sections, repeats, or something else?",
        "How did the schema handle modulations, inversions, ambiguous harmony, pickups, repeats, and transcription disagreement?",
        "Which labeling-tool decision most improved consistency or throughput?",
        "How was label quality checked?",
        "What training comparison supports the recognition improvement?",
        "Why was the SOTA statement subjective, and what objective benchmark was missing?"
      ]
    },
    {
      "id": "evaluation_suite",
      "title": "4. Evaluation suite, Suno comparison, and SongEval",
      "minimum_evidence": "The release decision flow, rater design, one QA mechanism, and correct public-benchmark attribution.",
      "questions": [
        "How were the more than ten subjective sets divided by capability, language, genre, and use case?",
        "Why did you use three-rater CMOS for routine releases and ten-rater MOS for competitor comparisons?",
        "How did you control order, loudness, rater fatigue, genre preference, and poor raters?",
        "How did you treat uncertainty and disagreement?",
        "What was personally designed or operated by you, and what was achieved by the model team?",
        "State the Suno v5.5 and SongEval claims in externally safe language, including what you are not claiming."
      ]
    },
    {
      "id": "consumer_feedback",
      "title": "5. Consumer feedback and 50% user growth",
      "minimum_evidence": "One feedback-to-evaluation loop, the attribution boundary, and explicit unknowns around causal measurement.",
      "questions": [
        "Which product behaviors or feedback were useful model-quality signals?",
        "How did you prevent popularity, exposure, novelty, or recommendation bias from masquerading as quality?",
        "How did product evidence enter benchmark or model-improvement priorities?",
        "What evidence supports saying that the work contributed to 50% user growth without claiming sole causality?",
        "Which parts of the growth analysis are unknown or owned by another function?",
        "Give me one complete, externally safe example of a product signal becoming an evaluation or model priority."
      ]
    },
    {
      "id": "reward_model",
      "title": "6. Outcome reward-model labeling",
      "minimum_evidence": "The task unit, rubric and QA loop, a precise agreement definition or explicit unknown, and personal ownership.",
      "questions": [
        "What exactly did one preference task ask raters to compare?",
        "How were conflicting musical dimensions represented in the rubric?",
        "How did you distinguish real taste disagreement from a broken task?",
        "What does nearly 90% human agreement mean operationally, and what does it not prove?",
        "How were difficult examples sampled across iterations?",
        "Which changes to the labeling system produced a measurable improvement?"
      ]
    },
    {
      "id": "automatic_evaluation",
      "title": "7. Automatic evaluation and LLM-as-judge",
      "minimum_evidence": "A dimension definition, gold-set construction, one calibration choice, one bias slice, and a clear statement that Yuchen evaluated rather than built aesthetic scorers.",
      "questions": [
        "Which text-side dimensions achieved more than 80% agreement with human gold?",
        "How were training or tuning examples separated from calibration or holdout examples?",
        "Which prompt or judge-design decisions did you personally make?",
        "How was disagreement inspected by slice rather than only in aggregate?",
        "How was pop bias in musicality or aesthetic scorers detected?",
        "What conditions kept humans as the release gold standard?"
      ]
    },
    {
      "id": "composition_representation",
      "title": "8. Data composition and representation decisions",
      "minimum_evidence": "Observation, competing hypotheses, intervention, validation, falsifiability, and no invented savings or stakeholder story.",
      "questions": [
        "For Latin grooves and blues vocals, what audible failure did you observe, and what alternatives did you consider before blaming data composition?",
        "What did the sampling-ratio or targeted-sourcing intervention actually change?",
        "Which evaluation showed recovery, and what regressions did you check?",
        "For the representation decision, what did higher information density and a lower quality ceiling mean in measurable terms?",
        "What evidence would have changed your mind?",
        "Which actor, timeline, cost, and saved-training-run details are undocumented and must remain unknown?"
      ]
    },
    {
      "id": "collaboration_engineering",
      "title": "9. Collaboration, leadership, and engineering boundary",
      "minimum_evidence": "The team operating model, one verified collaboration example or explicit unknown, concrete verification practice, and honest implementation limits.",
      "questions": [
        "Describe the real approximately ten-person operating structure without naming private individuals.",
        "How were instructions, calibration, escalation, retraining, and rater removal handled?",
        "Give one documented cross-functional decision with model researchers or product. If no exact story exists, say that clearly.",
        "Which Python, SQL, audio-DSP, Docker, CI, or PyTorch tasks can you independently specify, review, debug, and own today?",
        "What did agent-assisted development change in your workflow, and how did you verify generated work?",
        "Which infrastructure and distributed-training tasks require stronger engineering support?"
      ]
    }
  ]
}
