<?xml version="1.0" encoding="UTF-8"?>
<opml version="2.0">
  <head>
    <title>101-02 Hypothesis testing, effect sizes, power, and multiplicity</title>
    <ownerName>Integrated Medical Foundations</ownerName>
  </head>
  <body>
    <outline text="Hypothesis testing, effect sizes, power, and multiplicity">
      <outline text="Tests and p values">
        <outline text="Asks if data are incompatible with a model">
          <outline text="Null: no difference, no association, reference value"/>
        </outline>
        <outline text="Does not decide theory, importance, or bias"/>
        <outline text="p: chance of a result at least as incompatible">
          <outline text="Assumes the null model and all assumptions"/>
        </outline>
        <outline text="Not the probability the null is true">
          <outline text="Nor chance occurrence, nor replication success"/>
        </outline>
        <outline text="Small p: effect, trivial effect, violation, bias, selection"/>
      </outline>
      <outline text="Error rates and power">
        <outline text="Significance level: prespecified long-run threshold"/>
        <outline text="Type one error: rejecting a true null"/>
        <outline text="Type two error: not rejecting a false null"/>
        <outline text="Properties of procedures, not labels for one result"/>
        <outline text="Power depends on size, effect, variability, design">
          <outline text="Plan for an effect worth detecting"/>
          <outline text="Post hoc power restates the p value"/>
          <outline text="Low prior plausibility: fewer significant findings true"/>
        </outline>
        <outline text="One-sided only if opposite effect equals no effect">
          <outline text="Unexpected harm is rarely irrelevant"/>
          <outline text="Specify direction before seeing outcomes"/>
        </outline>
      </outline>
      <outline text="Effect sizes and clinical importance">
        <outline text="Statistical significance is not clinical significance"/>
        <outline text="Risk difference: absolute change"/>
        <outline text="Risk ratio: proportional change"/>
        <outline text="Odds ratio looks more extreme when events are common">
          <outline text="Approximates risk ratio when events are rare"/>
        </outline>
        <outline text="Mean difference keeps natural units"/>
        <outline text="Standardised difference trades meaning for comparability"/>
        <outline text="Minimal important difference: smallest worthwhile change">
          <outline text="Varies with severity, burden, cost, harms"/>
        </outline>
        <outline text="Compare interval with benefit, trivial, harm zones">
          <outline text="Significant but trivial, or nonsignificant yet important"/>
        </outline>
      </outline>
      <outline text="Matching analysis to data structure">
        <outline text="Paired analysis removes stable between-unit variation">
          <outline text="Independent test on paired data wastes information"/>
          <outline text="Paired test on unrelated data invents dependence"/>
        </outline>
        <outline text="Repeated measures need correlation over time">
          <outline text="Treating all as independent gives false precision"/>
        </outline>
        <outline text="Rank-based tests are not assumption-free">
          <outline text="Large samples: mean comparisons fairly robust"/>
        </outline>
        <outline text="Chi-square needs adequate expected counts"/>
        <outline text="Software p values are not self-validating">
          <outline text="Normality pretest makes unstable two-stage inference"/>
        </outline>
      </outline>
      <outline text="Multiplicity">
        <outline text="Many outcomes, times, subgroups, models, looks">
          <outline text="Chance of at least one false rejection rises"/>
        </outline>
        <outline text="Family-wise error: any false rejection"/>
        <outline text="False discovery rate: expected false proportion"/>
        <outline text="Bonferroni divides threshold by number of tests">
          <outline text="Conservative when tests correlate"/>
        </outline>
        <outline text="Stepwise procedures improve power"/>
        <outline text="No adjustment rescues hidden unsuccessful tests"/>
      </outline>
      <outline text="Selective reporting and subgroups">
        <outline text="Choices made after results are known">
          <outline text="Researcher degrees of freedom yield convincing p values"/>
        </outline>
        <outline text="Prespecification, registration, protocols, code"/>
        <outline text="Exploration valuable if labelled and retested"/>
        <outline text="Subgroup claims need a direct interaction test">
          <outline text="Not significant here and nonsignificant there"/>
        </outline>
        <outline text="Credible if prespecified, few, plausible, replicated"/>
        <outline text="Arbitrary cut points create apparent thresholds"/>
      </outline>
      <outline text="Equivalence and non-inferiority">
        <outline text="Reverse the usual burden"/>
        <outline text="Equivalence: whole interval within both margins"/>
        <outline text="Non-inferiority: exclude unacceptable loss">
          <outline text="Margin preserves justified comparator benefit"/>
        </outline>
        <outline text="Bias toward similarity can fake non-inferiority">
          <outline text="Poor adherence, outcome misclassification"/>
          <outline text="Examine intention-to-treat and per-protocol"/>
        </outline>
      </outline>
      <outline text="Reproducibility and publication bias">
        <outline text="Computational: same data and code, same result"/>
        <outline text="Replicability: new data, compatible evidence"/>
        <outline text="Generalisability: other populations and settings"/>
        <outline text="Not compressible into p below a threshold"/>
        <outline text="Publication bias favours positive, novel results">
          <outline text="Small-study effects from selective reporting"/>
          <outline text="Funnel asymmetry cannot diagnose one cause"/>
        </outline>
        <outline text="Pooling biased studies: precisely wrong"/>
      </outline>
      <outline text="Sequential evidence and interpretation">
        <outline text="Repeated looks inflate false positives">
          <outline text="Group-sequential boundaries, alpha spending"/>
        </outline>
        <outline text="Early stopping for benefit exaggerates effects"/>
        <outline text="Safety: weigh severity and pattern"/>
        <outline text="State effect, interval, model, comparison">
          <outline text="Then examine bias and relevance"/>
        </outline>
        <outline text="Convergent evidence changes understanding"/>
      </outline>
    </outline>
  </body>
</opml>
