{
  "datasetId": "ai-product-model-selection-scorecard-v1",
  "schemaVersion": "1.0.0",
  "createdAt": "2026-07-21",
  "language": "zh-CN",
  "license": "https://creativecommons.org/licenses/by/4.0/",
  "licenseScope": "The field structure, Chinese descriptions, decision states, and blank records created by DragonAI. Vendor materials, benchmark results, prices, and user data are excluded and retain their own rights and restrictions.",
  "purposeZh": "用于按同一任务、硬门槛、实验配置、质量失败、运行成本、供应治理和退出条件比较大模型候选。",
  "disclosureZh": "DragonAI 运营 AI 产品经理课程。本文件是第一方字段规范与空白记录，不包含真实模型跑分、厂商排名、价格、用户数据或采购结论，也不是统一行业标准。",
  "decisionStates": [
    {
      "state": "adopt",
      "criterionZh": "硬门槛和最终测试均通过，适用范围、责任、监测与回退已明确。"
    },
    {
      "state": "limited_pilot",
      "criterionZh": "仅允许在声明范围内试用，仍有未决风险或证据需验证。"
    },
    {
      "state": "hold",
      "criterionZh": "证据不足、外部条件未满足或需要补测，暂不扩大。"
    },
    {
      "state": "reject",
      "criterionZh": "硬门槛失败、严重风险不可接受或增益不足以覆盖成本。"
    }
  ],
  "sections": [
    {
      "sectionId": "task-contract",
      "nameZh": "任务合同与非模型基线",
      "requiredFields": [
        "taskId",
        "targetUsers",
        "trigger",
        "inputSources",
        "inputDistribution",
        "idealOutputAttributes",
        "requiredEvidence",
        "allowedTools",
        "prohibitedBehavior",
        "humanHandoff",
        "baseline",
        "successCriteria",
        "stopCriteria"
      ],
      "verificationQuestionsZh": [
        "是否先证明任务需要概率模型？",
        "候选是否使用同一任务与失败切片？"
      ]
    },
    {
      "sectionId": "eligibility-gates",
      "nameZh": "硬门槛",
      "requiredFields": [
        "gateId",
        "candidateId",
        "modality",
        "deployment",
        "region",
        "dataHandling",
        "licenseAndTerms",
        "capacity",
        "tooling",
        "identityAndAudit",
        "support",
        "owner",
        "checkedAt",
        "evidenceUrls",
        "gateStatus"
      ],
      "gateStatusValues": [
        "pass",
        "fail",
        "unknown"
      ],
      "verificationQuestionsZh": [
        "任何 fail 是否阻断软指标排序？",
        "未知条件是否保持 unknown 而非默认通过？"
      ]
    },
    {
      "sectionId": "candidate-identity",
      "nameZh": "候选身份",
      "requiredFields": [
        "candidateId",
        "provider",
        "modelId",
        "modelVersion",
        "endpointOrDeployment",
        "region",
        "interface",
        "observedAt",
        "configurationSnapshot",
        "differenceHypothesis"
      ],
      "verificationQuestionsZh": [
        "身份是否足以重放？",
        "滚动别名的不可冻结限制是否记录？"
      ]
    },
    {
      "sectionId": "experiment-control",
      "nameZh": "同口径实验控制",
      "requiredFields": [
        "experimentId",
        "candidateId",
        "configurationMode",
        "testPhase",
        "frozenAt",
        "datasetVersion",
        "split",
        "systemPromptVersion",
        "examplesVersion",
        "retrievalVersion",
        "toolSchemaVersion",
        "sampling",
        "timeout",
        "retryPolicy",
        "concurrency",
        "runCount",
        "changeLog"
      ],
      "configurationModeValues": [
        "common",
        "candidate_optimized"
      ],
      "testPhaseValues": [
        "development",
        "final_unseen"
      ],
      "verificationQuestionsZh": [
        "共同配置与候选优化配置是否分开？",
        "最终测试是否保持未见？"
      ]
    },
    {
      "sectionId": "quality-and-failures",
      "nameZh": "逐例质量、切片与严重失败",
      "requiredFields": [
        "caseId",
        "candidateId",
        "experimentId",
        "runId",
        "configurationSnapshotReference",
        "input",
        "outputReference",
        "toolOrCitationTrace",
        "labels",
        "slice",
        "severity",
        "blocker",
        "reviewer",
        "reviewMethod",
        "judgeModelId",
        "judgeModelVersion",
        "judgePromptVersion",
        "rubricVersion",
        "presentationOrder",
        "perCaseScore",
        "explanation",
        "humanGroundTruthReference",
        "humanLabel",
        "disagreementStatus",
        "result"
      ],
      "reviewMethodValues": [
        "human",
        "model_judge",
        "computed",
        "hybrid"
      ],
      "disagreementStatusValues": [
        "not_applicable",
        "agree",
        "disagree",
        "needs_adjudication"
      ],
      "resultValues": [
        "pass",
        "fail",
        "needs_review"
      ],
      "verificationQuestionsZh": [
        "严重失败是否单独展示？",
        "重复运行波动和裁判分歧是否保留？"
      ]
    },
    {
      "sectionId": "operations-and-cost",
      "nameZh": "运行性能与单位合格任务成本",
      "requiredFields": [
        "candidateId",
        "loadProfile",
        "window",
        "latencyPercentiles",
        "throughput",
        "errorRate",
        "retryRate",
        "inputUnits",
        "outputUnits",
        "toolAndRetrievalCost",
        "humanReviewCost",
        "reworkCost",
        "qualifiedTaskCost",
        "currency",
        "pricingSource",
        "observedAt"
      ],
      "verificationQuestionsZh": [
        "是否在同一负载窗口测量？",
        "成本是否包含重试、工具和人工返工？"
      ]
    },
    {
      "sectionId": "supplier-governance",
      "nameZh": "供应、数据与变更治理",
      "requiredFields": [
        "candidateId",
        "dataPath",
        "retentionAndDeletion",
        "monitoring",
        "versionChangeTrigger",
        "termsChangeTrigger",
        "capacityTrigger",
        "incidentOwner",
        "fallbackCandidate",
        "nonModelFallback",
        "reviewAfter"
      ],
      "verificationQuestionsZh": [
        "版本、条款和容量变化是否触发重测？",
        "回退路径是否已验证？"
      ]
    },
    {
      "sectionId": "selection-decision",
      "nameZh": "选型决定",
      "requiredFields": [
        "decisionId",
        "candidateId",
        "decisionState",
        "scope",
        "supportingRunIds",
        "failedOrUnknownGateIds",
        "blockers",
        "decisionReason",
        "unresolvedRisks",
        "owner",
        "decidedAt",
        "stopConditions",
        "fallback",
        "reviewTriggers"
      ],
      "verificationQuestionsZh": [
        "决定是否绑定明确版本与范围？",
        "硬门槛是否禁止被平均分抵消？"
      ]
    }
  ],
  "blankCandidate": {
    "candidateId": "",
    "provider": "",
    "modelId": "",
    "modelVersion": "",
    "endpointOrDeployment": "",
    "region": "",
    "interface": "",
    "observedAt": "",
    "configurationSnapshot": {},
    "differenceHypothesis": ""
  },
  "blankDecision": {
    "decisionId": "",
    "candidateId": "",
    "decisionState": "hold",
    "scope": "",
    "supportingRunIds": [],
    "failedOrUnknownGateIds": [],
    "blockers": [],
    "decisionReason": "",
    "unresolvedRisks": [],
    "owner": "",
    "decidedAt": "",
    "stopConditions": [],
    "fallback": "",
    "reviewTriggers": []
  },
  "sources": [
    {
      "evidenceId": "google-ml-problem-framing-20260720",
      "url": "https://developers.google.com/machine-learning/problem-framing/problem-framing"
    },
    {
      "evidenceId": "microsoft-foundry-model-benchmarks-20260721",
      "url": "https://learn.microsoft.com/en-us/azure/foundry/concepts/model-benchmarks"
    },
    {
      "evidenceId": "google-vertex-judge-model-evaluation-20260721",
      "url": "https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/evaluate-judge-model"
    },
    {
      "evidenceId": "nist-ai-rmf-genai-profile-20260721",
      "url": "https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-generative-artificial-intelligence"
    }
  ],
  "title": "AI 产品大模型选型证据卡 v1",
  "description": "按任务合同、硬门槛、候选身份、实验控制、逐例质量、运行成本、供应治理和退出条件记录大模型选型证据。",
  "url": "https://course.dragonai.tech/datasets/ai-product-model-selection-scorecard-v1.json",
  "creator": "烛龙智元内容研究组"
}
