{
  "schemaVersion": 1,
  "study": "cold-email-personalization-accuracy-study",
  "title": "Personalization accuracy study: study protocol",
  "status": "awaiting-real-data",
  "scope": "Collection template, not a dataset or completed field study. All null values are intentionally uncollected.",
  "unitOfObservation": "factual-claim",
  "requiredInputs": "Saved source excerpts, generation metadata and claim-level outputs.",
  "controls": "Hide model name during review; retain unverifiable claims separately.",
  "primaryAnalysis": "Supported, contradicted, unsupported and unverifiable claim counts; also aggregate by message.",
  "registration": {
    "owner": null,
    "protocolVersion": "1",
    "registeredAt": null,
    "population": null,
    "eligibilityRules": null,
    "primaryOutcomeDefinition": null,
    "denominator": null,
    "observationWindow": null,
    "targetSampleAndJustification": null,
    "assignmentProcedureAndSeed": null,
    "exclusionRules": null,
    "safetyStopConditions": null,
    "missingDataPolicy": null,
    "retentionAndAccessOwner": null
  },
  "collectionRules": [
    "Assign pseudonymous IDs; do not store mailbox passwords, access tokens or unnecessary identifiers.",
    "Freeze the outcome and exclusions before collection; retain exclusions with reasons.",
    "Do not use repeated observations as independent people. Record clustering by person, company or mailbox.",
    "Do not treat a software test, simulated reply or researcher-written example as participant or campaign evidence.",
    "Record complaints and opt-outs and follow the actual suppression workflow.",
    "For interviews obtain permission before recording and respect withdrawal.",
    "Publish aggregate or redacted evidence only after review; keep original private records outside the public website."
  ],
  "fieldDictionary": {
    "message_id": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "message id",
      "missingValue": null
    },
    "claim_text": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "claim text",
      "missingValue": null
    },
    "source_excerpt": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "source excerpt",
      "missingValue": null
    },
    "source_captured_at": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "source captured at",
      "missingValue": null
    },
    "prompt_version": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "prompt version",
      "missingValue": null
    },
    "model_version": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "model version",
      "missingValue": null
    },
    "review_label": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "review label",
      "missingValue": null
    },
    "evidence_reason": {
      "required": true,
      "type": "string, number or boolean as specified by the registered protocol; null until collected",
      "meaning": "evidence reason",
      "missingValue": null
    }
  },
  "blankRecord": {
    "observation_id": null,
    "message_id": null,
    "claim_text": null,
    "source_excerpt": null,
    "source_captured_at": null,
    "prompt_version": null,
    "model_version": null,
    "review_label": null,
    "evidence_reason": null,
    "excluded": false,
    "exclusion_reason": null,
    "protocol_deviation": null
  },
  "records": [],
  "analysisStatus": "not-run",
  "results": null,
  "interpretationLimits": "A fluent paraphrase is not evidence. Exclude inaccessible sources from verified counts and show them as unresolved. Publish results only after the evidence, method and limitations have been reviewed. This protocol provides no benchmark, expected lift or completed-study claim.",
  "procedure": [
    "Store the exact source excerpt, retrieval date, prompt version and generated claim. Separate facts from stylistic suggestions.",
    "Review claims against the saved source without seeing the preferred model name. Label supported, contradicted, unsupported or unverifiable.",
    "Before collection, write the primary outcome, observation window, exclusion rules and stopping conditions. Preserve excluded observations with a reason rather than quietly removing them.",
    "Pilot the procedure with fictional or owned test data, resolve ambiguous fields, and freeze a dated protocol version before the main run."
  ],
  "interviewPrompts": []
}
