{
  "schema_version": 1,
  "title": "Mathematical Foundations for Computer Science and AI",
  "planning_date": "2026-10-04",
  "status": "complete",
  "destination": "docs/tutorials/math",
  "specification": "PLAN.md",
  "languages": [
    "en",
    "zh-CN"
  ],
  "module_count": 32,
  "total_hours": 328,
  "supplementary_modules": [
    {
      "number": 0,
      "status": "available",
      "title": {
        "en": "Optional algebra and Python orientation",
        "zh-CN": "可选代数复习与 Python 入门"
      },
      "hours": 6,
      "additional_optional_setup_hours": 1
    }
  ],
  "taught_module_contract": {
    "session_count": 4,
    "concept_sections": 6,
    "minimum_worked_examples": 6,
    "labs": 3,
    "required_exercises": 12,
    "optional_extension_exercises": 2,
    "quiz_questions": 10
  },
  "parts": [
    {
      "id": "A",
      "title": "Mathematical language and reasoning",
      "modules": [
        1,
        2,
        3,
        4
      ],
      "hours": 32
    },
    {
      "id": "B",
      "title": "Discrete mathematics for CS",
      "modules": [
        5,
        6,
        7,
        8
      ],
      "hours": 34
    },
    {
      "id": "C",
      "title": "Linear algebra and geometry",
      "modules": [
        9,
        10,
        11,
        12,
        13,
        14
      ],
      "hours": 58
    },
    {
      "id": "D",
      "title": "Calculus and change",
      "modules": [
        15,
        16,
        17,
        18
      ],
      "hours": 40
    },
    {
      "id": "E",
      "title": "Optimisation",
      "modules": [
        19,
        20
      ],
      "hours": 24
    },
    {
      "id": "F",
      "title": "Probability and statistics",
      "modules": [
        21,
        22,
        23,
        24,
        25,
        26
      ],
      "hours": 64
    },
    {
      "id": "G",
      "title": "Information and learning",
      "modules": [
        27,
        28,
        29,
        30
      ],
      "hours": 46
    },
    {
      "id": "H",
      "title": "Integrated projects",
      "modules": [
        31,
        32
      ],
      "hours": 30
    }
  ],
  "routes": [
    {
      "id": "full",
      "title": "Full foundation",
      "modules": [
        1,
        2,
        3,
        4,
        5,
        6,
        7,
        8,
        9,
        10,
        11,
        12,
        13,
        14,
        15,
        16,
        17,
        18,
        19,
        20,
        21,
        22,
        23,
        24,
        25,
        26,
        27,
        28,
        29,
        30,
        31,
        32
      ],
      "branches": [
        "continuous_probability",
        "vector_covariance_extension"
      ],
      "hours": 328,
      "optional_extensions": [
        "svm_duality_extension"
      ]
    },
    {
      "id": "cs",
      "title": "CS core",
      "modules": [
        1,
        2,
        3,
        4,
        5,
        6,
        7,
        8,
        21,
        22,
        23,
        24,
        31
      ],
      "branches": [
        "discrete_probability"
      ],
      "hours": 120,
      "optional_extensions": []
    },
    {
      "id": "ai",
      "title": "AI foundation",
      "modules": [
        1,
        2,
        3,
        4,
        5,
        6,
        9,
        10,
        11,
        12,
        13,
        14,
        15,
        16,
        17,
        18,
        19,
        21,
        22,
        23,
        24,
        25,
        26,
        27,
        28,
        29,
        30,
        32
      ],
      "branches": [
        "continuous_probability",
        "vector_covariance_extension"
      ],
      "hours": 286,
      "optional_extensions": [
        "svm_duality_extension"
      ]
    },
    {
      "id": "ai_entry",
      "title": "Preparation for the existing AI series",
      "modules": [
        1,
        2,
        3,
        4,
        5,
        9,
        10,
        11,
        12,
        13,
        15,
        16,
        17,
        18,
        19,
        21,
        22,
        23,
        24,
        25
      ],
      "branches": [
        "continuous_probability",
        "vector_covariance_extension"
      ],
      "hours": 192,
      "optional_extensions": []
    }
  ],
  "modules": [
    {
      "number": 1,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Mathematical language numbers and functions",
        "zh-CN": "数学语言 数与函数"
      },
      "hours": 6,
      "prerequisites": [],
      "conditional_prerequisites": [],
      "prerequisite_description": "entry algebra skills or Module 00",
      "part": "A",
      "outcomes": [
        "distinguish definitions from claims",
        "track free and bound symbols",
        "state a function's domain and codomain",
        "expand a finite sum or product",
        "translate an expression into a small program"
      ],
      "lesson_sequence": [
        "Mathematical objects and number systems: natural numbers with an explicit zero convention, integers, rationals, reals, and a preview of complex numbers.",
        "Symbols, equality, inequalities, indexed notation, intervals, absolute value, and the difference between an assignment and an equation.",
        "Functions as mappings: domain, codomain, image, composition, inverse functions, and piecewise definitions.",
        "Powers and logarithms: domains, inverse relationships, growth comparisons, and logarithm base changes.",
        "Finite sums and products: indices, empty sums/products, double sums, and avoiding variable capture.",
        "Reading a mathematical specification: inputs, outputs, assumptions, and checking a proposed identity."
      ],
      "worked_examples": "evaluate a piecewise absolute-value function; compose `f(x) = 2x + 1` with `g(x) = x²`; expand `Σᵢ₌₀³ (2i + 1) = 16`; compare `log₂ n` and `ln n`; find where `log(x − 1)` is defined",
      "labs": [
        {
          "id": "A",
          "description": "implements finite sums and compares loops with written expansions"
        },
        {
          "id": "B",
          "description": "plots linear, quadratic, exponential, and logarithmic functions over stated domains"
        },
        {
          "id": "C",
          "description": "repairs an off-by-one sum and an incorrect assumption that a general function distributes over addition"
        }
      ],
      "exit_assessment": "specify and implement a piecewise function, explain all symbols in a double sum, and reject `log(a + b) = log a + log b` with a valid counterexample",
      "visual": "function composition explorer with domain warnings",
      "connections": "program contracts and indexed AI losses",
      "misconception": "treating a codomain as the set of values actually attained",
      "reading_keys": [
        "R1",
        "R6"
      ],
      "reading_description": "R1 definitions and functions; R6 introductory notation"
    },
    {
      "number": 2,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Logic quantifiers and specifications",
        "zh-CN": "逻辑 量词与规格说明"
      },
      "hours": 8,
      "prerequisites": [
        1
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01",
      "part": "A",
      "outcomes": [
        "construct truth tables",
        "distinguish implication from its converse",
        "negate quantified statements",
        "translate requirements into predicates",
        "identify vacuous truth and scope errors"
      ],
      "lesson_sequence": [
        "Propositions and Boolean connectives, with truth-table semantics.",
        "Implication, equivalence, necessary/sufficient conditions, converse, inverse, and contrapositive.",
        "Predicates, universal and existential quantifiers, and finite versus infinite domains.",
        "Quantifier scope, nested order, and negation using De Morgan's laws.",
        "Preconditions, postconditions, and assertions for simple programs.",
        "Satisfiability, validity, and finite model checking; explain what enumeration can establish."
      ],
      "worked_examples": "compare `∀x ∃y` with `∃y ∀x` using integer addition; negate “every request has an approved reviewer”; derive the contrapositive of “divisible by four implies even”; evaluate an implication with a false antecedent",
      "labs": [
        {
          "id": "A",
          "description": "builds truth tables from small expressions"
        },
        {
          "id": "B",
          "description": "checks quantified predicates over a finite catalogue"
        },
        {
          "id": "C",
          "description": "generates counterexamples to a wrongly formalised access rule"
        }
      ],
      "exit_assessment": "translate three requirements with explicit domains, prove two truth-table equivalences, and explain why a finite search is insufficient for a universal claim over all integers",
      "visual": "truth-table and quantifier explorer",
      "connections": "branching logic, verification, and the assumptions behind universal AI claims",
      "misconception": "confusing “if” with “if and only if”",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 propositions and predicate logic"
    },
    {
      "number": 3,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Sets relations and discrete structures",
        "zh-CN": "集合 关系与离散结构"
      },
      "hours": 8,
      "prerequisites": [
        1,
        2
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01 and 02",
      "part": "A",
      "outcomes": [
        "compute set operations",
        "represent and classify relations",
        "build equivalence classes",
        "distinguish an order from an equivalence",
        "test injectivity and surjectivity"
      ],
      "lesson_sequence": [
        "Membership, subsets, power sets, complements relative to a universe, and Cartesian products.",
        "Set identities and their proof by membership or logical equivalence.",
        "Binary relations and reflexive, symmetric, antisymmetric, and transitive properties.",
        "Equivalence relations, partitions, and quotient sets through concrete examples.",
        "Partial and total orders, Hasse diagrams, and incomparable elements.",
        "Functions as relations; injective, surjective, bijective, and inverse mappings; a brief countability preview."
      ],
      "worked_examples": "form all subsets of `{a, b, c}`; partition integers by parity; order sets by inclusion; explain why “shares a keyword” is generally not transitive; compare an injective identifier mapping with a lossy feature mapping",
      "labs": [
        {
          "id": "A",
          "description": "checks set identities on small universes"
        },
        {
          "id": "B",
          "description": "classifies relations represented as pairs or Boolean tables"
        },
        {
          "id": "C",
          "description": "groups records under an equivalence relation and exposes a broken deduplication rule"
        }
      ],
      "exit_assessment": "prove one set identity, give a counterexample for each failed relation property, and construct a Hasse diagram",
      "visual": "relation explorer with highlighted failing pairs/triples",
      "connections": "databases, dependency orders, equivalence of representations, and data labels",
      "misconception": "interpreting antisymmetric as “never symmetric”",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 sets, relations, and mappings"
    },
    {
      "number": 4,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Proof methods induction and invariants",
        "zh-CN": "证明方法 归纳法与不变式"
      },
      "hours": 10,
      "prerequisites": [
        2,
        3
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "02 and 03",
      "part": "A",
      "outcomes": [
        "choose an appropriate proof method",
        "write complete induction arguments",
        "distinguish a counterexample from a proof",
        "prove correctness and termination separately",
        "state a loop invariant at the correct program point"
      ],
      "lesson_sequence": [
        "Theorems, hypotheses, conclusions, lemmas, and the role of definitions.",
        "Direct proof, cases, contrapositive, and contradiction.",
        "Existence and uniqueness proofs; constructive arguments and counterexamples.",
        "Weak and strong induction, base cases, induction hypotheses, and well-ordering.",
        "Recursive definitions and structural induction over lists and trees.",
        "Loop invariants, initialisation, preservation, postconditions, and decreasing termination measures."
      ],
      "worked_examples": "prove a sum formula; prove a parity claim by contrapositive; use strong induction on a recursive process; verify linear search returns the first matching position; explain why its invariant alone does not prove termination",
      "labs": [
        {
          "id": "A",
          "description": "exhaustively checks small instances of a proposed identity while recording that this is experimental evidence"
        },
        {
          "id": "B",
          "description": "instruments a search invariant"
        },
        {
          "id": "C",
          "description": "finds the missing base case and non-decreasing variant in two faulty arguments"
        }
      ],
      "exit_assessment": "submit one direct proof, one induction proof, and one correctness/termination argument with separate obligations",
      "visual": "execution trace with invariant and variant annotations",
      "connections": "algorithm correctness, recursive definitions, and reading mathematical AI derivations",
      "misconception": "using the statement to be proved as an intermediate assumption",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 proof methods, induction, and state machines"
    },
    {
      "number": 5,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Counting combinatorics and finite probability",
        "zh-CN": "计数 组合数学与有限概率"
      },
      "hours": 8,
      "prerequisites": [
        3,
        4
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "03 and 04",
      "part": "B",
      "outcomes": [
        "choose sum/product rules correctly",
        "count ordered and unordered selections",
        "apply inclusion–exclusion and the pigeonhole principle",
        "identify when equally likely counting gives a probability"
      ],
      "lesson_sequence": [
        "Addition and multiplication rules, decision trees, and disjointness assumptions.",
        "Permutations with and without repeated objects.",
        "Combinations, binomial coefficients, Pascal's identity, and the binomial theorem.",
        "Counting with repetition, stars and bars, and constraints on selections.",
        "Inclusion–exclusion, complementary counting, and the pigeonhole principle.",
        "Finite sample spaces, equally likely outcomes, and a preview of probability versus counting."
      ],
      "worked_examples": "count length-four binary strings; choose three records from ten; distribute five identical tasks across three named workers; count records matching either of two labels; compute the probability of at least one collision in a small uniform hash space",
      "labs": [
        {
          "id": "A",
          "description": "enumerates small selections and checks counting formulas"
        },
        {
          "id": "B",
          "description": "compares exact collision counts with a simulation"
        },
        {
          "id": "C",
          "description": "repairs a count that treats repeated labels as distinct outcomes"
        }
      ],
      "exit_assessment": "explain the modelling assumptions before each count; prove Pascal's identity combinatorially; calculate a collision probability without confusing it with a guarantee",
      "visual": "selection tree switching between order-sensitive and order-insensitive tasks",
      "connections": "search spaces, hash collisions, and dataset splits",
      "misconception": "assuming a finite sample space is automatically uniform",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 counting and combinatorial probability"
    },
    {
      "number": 6,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Sequences sums asymptotics and recurrences",
        "zh-CN": "数列 求和 渐近分析与递推关系"
      },
      "hours": 10,
      "prerequisites": [
        1,
        4
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01 and 04",
      "part": "B",
      "outcomes": [
        "evaluate common finite sums",
        "prove Big O and Theta statements from definitions",
        "solve simple recurrences",
        "distinguish exact operation counts from asymptotic growth and elapsed time"
      ],
      "lesson_sequence": [
        "Arithmetic and geometric sequences, finite sums, and telescoping.",
        "Growth rates, logarithm bases, and comparison of polynomials and exponentials.",
        "Big O, Omega, Theta, little o, quantified definitions, and witnesses.",
        "Operation counts for loops, nested loops, and input-dependent branches.",
        "Recurrences solved by expansion, recursion trees, and induction/substitution.",
        "Master-theorem cases with stated applicability; simple aggregate amortised analysis."
      ],
      "worked_examples": "sum `1 + … + n`; count a triangular loop; prove `3n² + 7n + 2` is Theta of `n²`; solve `T(n) = 2T(n/2) + n` for powers of two; analyse doubling capacity using total copies",
      "labs": [
        {
          "id": "A",
          "description": "records exact counts for three loops"
        },
        {
          "id": "B",
          "description": "compares recurrence counts against predicted growth"
        },
        {
          "id": "C",
          "description": "diagnoses misleading timings and a recurrence outside the standard Master theorem"
        }
      ],
      "exit_assessment": "provide explicit constants for a Big O proof, solve and verify a recurrence, and distinguish worst-case from amortised cost",
      "visual": "growth and recurrence-tree explorer",
      "connections": "algorithm analysis and the cost of matrix-based AI computation",
      "misconception": "interpreting Big O as an exact time or a tight bound",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 sums, asymptotics, and recurrences"
    },
    {
      "number": 7,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Graphs trees and state transitions",
        "zh-CN": "图 树与状态转移"
      },
      "hours": 8,
      "prerequisites": [
        3,
        4,
        6
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "03, 04, and 06",
      "part": "B",
      "outcomes": [
        "choose a graph model",
        "prove a basic graph property",
        "explain BFS and DFS invariants",
        "identify DAGs and valid topological orders",
        "relate a state-transition graph to reachability"
      ],
      "lesson_sequence": [
        "Directed/undirected graphs, vertices, edges, degrees, paths, walks, and representations.",
        "Degree sums, connectivity, cycles, and simple counting arguments.",
        "Trees, rooted trees, leaves, height, and the edge-count theorem.",
        "BFS/DFS as mathematical traversals; reachability and unweighted shortest paths.",
        "DAGs, topological ordering, dependency scheduling, and induction over a graph.",
        "Finite state machines, product states, invariants, and limits of exhaustive exploration."
      ],
      "worked_examples": "prove a finite tree with `n` vertices has `n − 1` edges; explain a BFS distance certificate; find a cycle in prerequisites; model a two-state retry protocol",
      "labs": [
        {
          "id": "A",
          "description": "builds adjacency lists and matrices"
        },
        {
          "id": "B",
          "description": "traces traversals and checks distance/order claims"
        },
        {
          "id": "C",
          "description": "explores a finite protocol and repairs an unreachable or unsafe state"
        }
      ],
      "exit_assessment": "prove the tree theorem, construct a topological order or a cycle witness, and explain BFS correctness in an unweighted graph",
      "visual": "graph traversal and state-space explorer",
      "connections": "dependency systems, automata, computation graphs, and graph data",
      "misconception": "expecting BFS to solve arbitrary weighted shortest-path problems",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 graphs and state machines"
    },
    {
      "number": 8,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Modular arithmetic and algebra for computing",
        "zh-CN": "模运算与计算中的代数"
      },
      "hours": 8,
      "prerequisites": [
        1,
        4
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01 and 04",
      "part": "B",
      "outcomes": [
        "work with divisibility and congruence",
        "compute gcd and modular inverses",
        "explain when cancellation is valid",
        "distinguish integers, modular rings, and prime finite fields"
      ],
      "lesson_sequence": [
        "Divisibility, primes, factorisation, and the division algorithm.",
        "Euclid's algorithm and its invariant; extended Euclid and Bézout coefficients.",
        "Congruence, modular addition/multiplication, and representative choices.",
        "Modular inverses, solvability of a linear congruence, and zero divisors.",
        "Fast modular exponentiation; Fermat's little theorem with conditions; a small Chinese remainder example.",
        "Groups, rings, and fields through examples; binary arithmetic, hashing, and a conceptual cryptography connection."
      ],
      "worked_examples": "compute `gcd(252, 105) = 21`; find the inverse of `3 mod 7`; explain why `2 mod 6` has no inverse; solve two coprime congruences; calculate a small exponent by repeated squaring",
      "labs": [
        {
          "id": "A",
          "description": "implements Euclid and verifies Bézout identities"
        },
        {
          "id": "B",
          "description": "implements modular powers and compares with Python's built-in result"
        },
        {
          "id": "C",
          "description": "detects illegal modular division and constructs a tiny checksum collision"
        }
      ],
      "exit_assessment": "prove gcd preservation, justify an inverse using gcd, and explain how a checksum differs from a cryptographic security claim",
      "visual": "residue clock and Euclidean trace",
      "connections": "arithmetic representations, checksums, hashes, and later finite-field study",
      "misconception": "dividing by a non-zero residue that is not invertible",
      "reading_keys": [
        "R1"
      ],
      "reading_description": "R1 number theory"
    },
    {
      "number": 9,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Vectors geometry and array notation",
        "zh-CN": "向量 几何与数组记法"
      },
      "hours": 8,
      "prerequisites": [
        1,
        3
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01 and 03",
      "part": "C",
      "outcomes": [
        "compute vector operations and distances",
        "interpret a dot product geometrically",
        "choose and explain a norm",
        "distinguish vectors from their coordinate arrays",
        "track shapes in simple data operations"
      ],
      "lesson_sequence": [
        "Scalars, vectors, coordinates, features, and points versus displacements.",
        "Addition, scalar multiplication, linear combinations, lines, and affine combinations.",
        "Dot products, lengths, angles, orthogonality, and Cauchy–Schwarz.",
        "Euclidean, Manhattan, and maximum norms; metric axioms and feature scales.",
        "Unit vectors, normalisation, cosine similarity, and the zero-vector exception.",
        "NumPy shape notation, row/column conventions, broadcasting, and higher-order arrays as a preview."
      ],
      "worked_examples": "calculate the distance between `(1, 2)` and `(4, 6)`; normalise `(3, 4)`; project `(2, 3)` onto a coordinate direction; show how changing a feature's units changes a nearest neighbour",
      "labs": [
        {
          "id": "A",
          "description": "implements dot products and norms with loops, then checks NumPy results"
        },
        {
          "id": "B",
          "description": "compares neighbour rankings under feature scaling"
        },
        {
          "id": "C",
          "description": "repairs accidental broadcasting and a division by zero in cosine similarity"
        }
      ],
      "exit_assessment": "justify a norm calculation, check all array shapes, and explain when two data vectors being close is meaningful",
      "visual": "draggable two-dimensional vectors with angle and norm displays",
      "connections": "graphics, embeddings, nearest neighbours, and attention scores",
      "misconception": "treating similarity as a probability or assuming all feature units are interchangeable",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 geometric introductions; R6 analytic geometry"
    },
    {
      "number": 10,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Matrices linear maps and linear systems",
        "zh-CN": "矩阵 线性映射与线性方程组"
      },
      "hours": 10,
      "prerequisites": [
        9
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "09",
      "part": "C",
      "outcomes": [
        "multiply compatible matrices",
        "interpret composition of linear maps",
        "solve small systems by elimination",
        "distinguish unique, inconsistent, and underdetermined systems",
        "use a solver instead of explicitly forming an inverse"
      ],
      "lesson_sequence": [
        "Matrices as linear transformations and data tables; columns as images of basis vectors.",
        "Products, transpose, identity, diagonal and block matrices, and non-commutativity.",
        "Linear systems, augmented matrices, elementary row operations, and pivots.",
        "Gaussian elimination, row-echelon forms, consistency, and free variables.",
        "Invertibility, determinants as signed volume scaling, and why determinant-based solving is unsuitable as a default numerical method.",
        "Batched linear models, affine maps with bias, and shape-safe implementation."
      ],
      "worked_examples": "compose a shear and a scale in both orders; solve a two-equation system; construct inconsistent and infinitely soluble systems; explain a `batch × features` matrix multiplied by `features × outputs` weights",
      "labs": [
        {
          "id": "A",
          "description": "implements small matrix products"
        },
        {
          "id": "B",
          "description": "traces elimination with pivoting for a small system"
        },
        {
          "id": "C",
          "description": "compares a direct solver with inverse-based code and diagnoses a singular matrix"
        }
      ],
      "exit_assessment": "solve and classify three systems, predict a transformed grid, and annotate every dimension in a batched affine layer",
      "visual": "matrix transformation explorer",
      "connections": "constraint systems, image transforms, dense layers, and attention products",
      "misconception": "assuming matrix multiplication commutes",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 systems and matrix multiplication; R6 linear algebra"
    },
    {
      "number": 11,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Vector spaces bases rank and identifiability",
        "zh-CN": "向量空间 基 秩与可辨识性"
      },
      "hours": 10,
      "prerequisites": [
        4,
        10
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "04 and 10",
      "part": "C",
      "outcomes": [
        "test linear independence",
        "find a basis and coordinates",
        "compute rank and nullity",
        "explain the four fundamental subspaces",
        "relate non-unique parameters to null-space directions"
      ],
      "lesson_sequence": [
        "Vector spaces and subspaces; closure and examples involving functions or polynomials.",
        "Span, dependence, independence, and redundant representations.",
        "Bases, dimension, coordinate changes, and uniqueness of coordinates.",
        "Column space, row space, null space, and left null space.",
        "Rank–nullity with a complete finite-dimensional argument; system solvability revisited.",
        "Identifiability: duplicate features, one-hot columns with an intercept, and equivalent parameter settings."
      ],
      "worked_examples": "find a basis for three vectors in two dimensions; compute a null space by elimination; show why Celsius and Fahrenheit columns plus an intercept are dependent; exhibit different weights with identical predictions",
      "labs": [
        {
          "id": "A",
          "description": "finds rank and null-space examples with exact small integers before using numerical tolerances"
        },
        {
          "id": "B",
          "description": "checks predictions after adding a null-space vector to weights"
        },
        {
          "id": "C",
          "description": "diagnoses redundant features and explains tolerance-dependent numerical rank"
        }
      ],
      "exit_assessment": "prove a set is a subspace, compute a basis for a null space, and explain the difference between unique predictions and unique parameters",
      "visual": "span and dependence explorer",
      "connections": "constraint solving, feature redundancy, and parameter identifiability",
      "misconception": "treating a nearly dependent matrix as exactly dependent without discussing precision",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 vector spaces and fundamental subspaces; R6 linear algebra"
    },
    {
      "number": 12,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Orthogonality projections and least squares",
        "zh-CN": "正交性 投影与最小二乘"
      },
      "hours": 10,
      "prerequisites": [
        11
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "11",
      "part": "C",
      "outcomes": [
        "compute an orthogonal projection",
        "construct an orthonormal basis",
        "derive normal equations geometrically",
        "solve least squares with QR",
        "explain residual orthogonality and rank-deficient solutions"
      ],
      "lesson_sequence": [
        "Inner products and orthogonality; weighted inner products as an extension.",
        "Projection onto a vector and onto a subspace.",
        "Orthonormal bases, Gram–Schmidt, and its numerical limitations.",
        "QR factorisation and a stable least-squares solve.",
        "Derive `Xᵀ(Xw − y) = 0` from projection; uniqueness conditions and the later calculus derivation.",
        "Minimum-norm solutions and the geometric role of a pseudoinverse, preparing for SVD."
      ],
      "worked_examples": "project `(3, 2)` onto the line spanned by `(1, 1)`; fit a line to three points; verify `Xᵀr = 0`; show why normal equations can magnify conditioning problems",
      "labs": [
        {
          "id": "A",
          "description": "calculates projections and residuals"
        },
        {
          "id": "B",
          "description": "fits a small linear model using QR and a library least-squares solver"
        },
        {
          "id": "C",
          "description": "exposes an unstable normal-equation fit using nearly collinear features"
        }
      ],
      "exit_assessment": "derive the normal equations without differentiation, explain every term's shape, and give conditions for a unique minimiser",
      "visual": "points, fit, and orthogonal residual explorer",
      "connections": "approximation, regression, and linear model baselines",
      "misconception": "concluding a small residual guarantees reliable coefficients or extrapolation",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 orthogonality and least squares; R6 analytic geometry and regression"
    },
    {
      "number": 13,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Eigenvalues spectral geometry and quadratic forms",
        "zh-CN": "特征值 谱几何与二次型"
      },
      "hours": 10,
      "prerequisites": [
        12
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "12",
      "part": "C",
      "outcomes": [
        "compute eigenpairs in small cases",
        "distinguish diagonalisation from the symmetric spectral theorem",
        "classify a symmetric quadratic form",
        "interpret repeated linear transformations",
        "connect spectra with conditioning"
      ],
      "lesson_sequence": [
        "Eigenvectors as invariant directions; characteristic equations in two dimensions.",
        "Diagonalisation, defective matrices, and the role of complex eigenvalues.",
        "Real symmetric spectral theorem and orthonormal eigenbases.",
        "Quadratic forms, positive semidefinite/definite matrices, and level-set geometry.",
        "Powers, spectral radius, and stability of a discrete linear iteration.",
        "Rayleigh quotients, principal directions, and condition ratios for positive-definite matrices."
      ],
      "worked_examples": "diagonalise a symmetric `2 × 2` matrix; contrast a rotation with a defective shear-like matrix; draw an ellipse from a positive-definite quadratic; predict convergence of `eₖ₊₁ = Beₖ`",
      "labs": [
        {
          "id": "A",
          "description": "compares hand eigenpairs with numerical ones"
        },
        {
          "id": "B",
          "description": "iterates a small linear map and predicts stable/unstable behaviour"
        },
        {
          "id": "C",
          "description": "diagnoses arbitrary eigenvector sign changes and the failure of diagonalisation"
        }
      ],
      "exit_assessment": "verify an eigenpair, classify three quadratic forms, and explain stability using the relevant spectral condition and matrix assumptions",
      "visual": "eigenvector and quadratic-contour explorer",
      "connections": "dynamical systems, graph spectra, optimisation curvature, and covariance geometry",
      "misconception": "assuming every matrix has a real orthonormal eigenbasis",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 eigenvalues and positive-definite matrices; R6 matrix decompositions"
    },
    {
      "number": 14,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "SVD low rank approximation and PCA",
        "zh-CN": "奇异值分解 低秩近似与主成分分析"
      },
      "hours": 10,
      "prerequisites": [
        13
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "13",
      "part": "C",
      "outcomes": [
        "interpret the shapes and geometry of an SVD",
        "compute a truncated approximation",
        "explain a pseudoinverse",
        "carry out centred PCA",
        "distinguish reconstruction error from predictive usefulness"
      ],
      "lesson_sequence": [
        "Singular values, left/right singular vectors, and full versus reduced SVD.",
        "Rotation/reflection–scaling–rotation geometry for rectangular matrices.",
        "Rank, pseudoinverse, minimum-norm solutions, and small singular values.",
        "Truncation and the best rank-k approximation theorem, with a worked error calculation.",
        "PCA by centring a data matrix and taking its SVD; sample scatter/covariance introduced algebraically.",
        "Explained variance, scaling choices, sign ambiguity, and leakage from fitting transformations on test data."
      ],
      "worked_examples": "write the SVD shapes for a `5 × 3` data matrix; reconstruct a rank-one approximation; relate squared singular values to centred scatter eigenvalues; compare PCA before and after changing feature units",
      "labs": [
        {
          "id": "A",
          "description": "compresses a generated matrix and measures reconstruction error"
        },
        {
          "id": "B",
          "description": "implements PCA using training-set centring"
        },
        {
          "id": "C",
          "description": "diagnoses an uncentred or test-fitted transform and an unstable pseudoinverse"
        }
      ],
      "exit_assessment": "derive the squared Frobenius reconstruction error from discarded singular values and explain the centring/scaling choices",
      "visual": "rank slider with reconstruction and error",
      "connections": "compression, dimensionality reduction, embeddings, and low-rank adaptation",
      "misconception": "interpreting high explained variance as evidence of good classification",
      "reading_keys": [
        "R2",
        "R6"
      ],
      "reading_description": "R2 SVD; R6 matrix decompositions and PCA. Probability interpretations return in Module 23"
    },
    {
      "number": 15,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Limits continuity and convergence",
        "zh-CN": "极限 连续性与收敛"
      },
      "hours": 8,
      "prerequisites": [
        1,
        4
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "01 and 04",
      "part": "D",
      "outcomes": [
        "reason about finite/infinite limits",
        "distinguish a sequence from its limit",
        "explain continuity",
        "identify divergent behaviour",
        "use convergence conditions rather than relying on a plot"
      ],
      "lesson_sequence": [
        "Sequences, convergence, boundedness, and monotone convergence examples.",
        "Function limits, one-sided limits, and algebraic limit laws with conditions.",
        "Epsilon–delta meaning through a linear example; separate intuition from proof.",
        "Continuity, composition, discontinuities, and intermediate/extreme value theorems on appropriate domains.",
        "Infinite geometric series and a small selection of comparison/ratio tests.",
        "Convergence of an iteration, tolerances, and what a finite numerical trace cannot establish."
      ],
      "worked_examples": "evaluate a removable singularity; show a geometric series converges only for an appropriate ratio; contrast `1/n` with `(-1)^n`; explain why a discontinuous threshold function has no derivative at its jump",
      "labs": [
        {
          "id": "A",
          "description": "samples sequences and compares conjectures with arguments"
        },
        {
          "id": "B",
          "description": "approaches a limit from both sides"
        },
        {
          "id": "C",
          "description": "constructs misleading finite plots and a falsely converged iteration"
        }
      ],
      "exit_assessment": "justify one limit with inequalities, state a theorem's hypotheses, and identify three failures of convergence",
      "visual": "limit and tolerance explorer",
      "connections": "iterative algorithms, approximation, and later optimisation convergence",
      "misconception": "treating “very close” as equal or assuming boundedness implies convergence",
      "reading_keys": [
        "R3"
      ],
      "reading_description": "R3 limits and infinite processes"
    },
    {
      "number": 16,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Derivatives Taylor approximation and sensitivity",
        "zh-CN": "导数 泰勒近似与敏感性"
      },
      "hours": 10,
      "prerequisites": [
        15
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "15",
      "part": "D",
      "outcomes": [
        "derive a derivative from a difference quotient",
        "use product/quotient/chain rules",
        "interpret local sensitivity",
        "construct a Taylor approximation with an error statement",
        "classify simple stationary points"
      ],
      "lesson_sequence": [
        "Derivatives as local linear approximations and rates of change.",
        "Derivatives of powers, exponentials, logarithms, sine, and cosine, with domains.",
        "Product, quotient, chain, and inverse-function rules.",
        "First/second derivatives, stationary points, mean value theorem, and local versus global extrema.",
        "Taylor polynomials, remainder conditions, and approximation error.",
        "Finite differences and the truncation/rounding tradeoff; non-differentiable absolute value and ReLU."
      ],
      "worked_examples": "differentiate a nested exponential; approximate `ln(1 + x)` near zero; show a stationary point need not be a minimum; calculate a central difference at several step sizes",
      "labs": [
        {
          "id": "A",
          "description": "compares analytic derivatives with difference quotients"
        },
        {
          "id": "B",
          "description": "explores Taylor error as the expansion point moves"
        },
        {
          "id": "C",
          "description": "diagnoses a gradient estimate at a kink and one using an excessively small step"
        }
      ],
      "exit_assessment": "derive and explain a chain-rule calculation, bound a small Taylor remainder under stated conditions, and classify a stationary point",
      "visual": "tangent and Taylor-order explorer",
      "connections": "sensitivity, root finding, and scalar training objectives",
      "misconception": "assuming every useful function is differentiable everywhere",
      "reading_keys": [
        "R3",
        "R6"
      ],
      "reading_description": "R3 differentiation and Taylor series; R6 vector calculus introduction"
    },
    {
      "number": 17,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Integration accumulation and simple differential equations",
        "zh-CN": "积分 累积与简单微分方程"
      },
      "hours": 10,
      "prerequisites": [
        16
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "16",
      "part": "D",
      "outcomes": [
        "interpret a definite integral",
        "apply the fundamental theorem of calculus with conditions",
        "normalise a simple density",
        "approximate an integral numerically",
        "solve and discretise a simple first-order ODE"
      ],
      "lesson_sequence": [
        "Riemann sums, signed area, accumulation, and the distinction between a definite integral and an antiderivative.",
        "Fundamental theorem of calculus and basic antiderivatives.",
        "Substitution and integration by parts through selected examples.",
        "Improper integrals, normalisation, and integration for continuous probability.",
        "Trapezoidal/Simpson quadrature and error behaviour on smooth functions.",
        "Separable first-order ODEs, exponential growth/decay, and Euler discretisation with stability warnings."
      ],
      "worked_examples": "compute an accumulated rate; normalise `p(x) = c x` on `[0, 1]`; find its mean; solve `dx/dt = −ax`; compare continuous decay with discrete Euler updates",
      "labs": [
        {
          "id": "A",
          "description": "implements quadrature and compares with an analytic integral"
        },
        {
          "id": "B",
          "description": "checks a candidate density's total mass"
        },
        {
          "id": "C",
          "description": "diagnoses unstable Euler steps and a divergent improper integral"
        }
      ],
      "exit_assessment": "compute an integral with a justified substitution, explain density normalisation, and state when an Euler simulation disagrees with a stable differential equation",
      "visual": "area partitions and decay-step explorer",
      "connections": "continuous uncertainty, numerical simulation, and dynamics used in later AI study",
      "misconception": "treating density height as probability",
      "reading_keys": [
        "R3"
      ],
      "reading_description": "R3 integration and differential equations"
    },
    {
      "number": 18,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Multivariable derivatives matrix calculus and automatic differentiation",
        "zh-CN": "多元微分 矩阵微积分与自动微分"
      },
      "hours": 12,
      "prerequisites": [
        10,
        16
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "10 and 16",
      "part": "D",
      "outcomes": [
        "compute partial/directional derivatives",
        "distinguish a gradient, Jacobian, and Hessian",
        "propagate derivatives through a computation graph",
        "derive matrix gradients with shapes",
        "verify a manual backward pass"
      ],
      "lesson_sequence": [
        "Functions of several variables, partial derivatives, level sets, and directional derivatives.",
        "Total differentials and gradients; differentiability versus existence of partial derivatives.",
        "Jacobians, composition, and the multivariable chain rule.",
        "Hessians, second-order Taylor expansions, and mixed-partial conditions.",
        "Derive gradients of dot products, quadratic forms, and least-squares losses using differentials.",
        "Forward/reverse automatic differentiation, vector–Jacobian products, shared nodes, and nondifferentiable conventions."
      ],
      "worked_examples": "derive `∇w ||Xw − y||² = 2Xᵀ(Xw − y)`; calculate a two-input nonlinear Jacobian; backpropagate through a repeated variable; distinguish the gradients of a summed and mean batch loss",
      "labs": [
        {
          "id": "A",
          "description": "computes and checks vector derivatives with finite differences"
        },
        {
          "id": "B",
          "description": "implements a tiny scalar reverse-mode engine"
        },
        {
          "id": "C",
          "description": "verifies a two-layer network backward pass and repairs a transpose, missing accumulation, or batch-factor fault"
        }
      ],
      "exit_assessment": "annotate every derivative's shape, derive a quadratic gradient, and trace one vector–Jacobian product by hand",
      "visual": "computation graph with forward values and reverse adjoints",
      "connections": "backpropagation, differentiable programs, and sensitivity analysis",
      "misconception": "assuming automatic differentiation is finite differencing or silently choosing a row-gradient convention",
      "reading_keys": [
        "R4",
        "R6"
      ],
      "reading_description": "R4 partial derivatives and chain rules; R6 vector calculus"
    },
    {
      "number": 19,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Convexity gradient methods and unconstrained optimisation",
        "zh-CN": "凸性 梯度方法与无约束优化"
      },
      "hours": 12,
      "prerequisites": [
        13,
        18
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "13 and 18",
      "part": "E",
      "outcomes": [
        "distinguish convexity from smoothness",
        "classify stationary points",
        "derive gradient descent on a quadratic",
        "state convergence assumptions",
        "compare gradient, Newton, and coordinate methods"
      ],
      "lesson_sequence": [
        "Objectives, feasible domains, minimisers, infima, and local/global optima.",
        "Convex sets/functions, Jensen's inequality, and first/second-order characterisations where applicable.",
        "Stationary conditions and Hessian classification; saddle points and flat directions.",
        "Gradient descent, learning rates, line search, and descent under a Lipschitz-gradient condition.",
        "Quadratic convergence analysis: eigenmodes, `0 < η < 2/λmax` for positive-definite curvature, and conditioning.",
        "Newton and coordinate descent; solve the Newton system rather than invert a Hessian; failure cases for nonconvex problems."
      ],
      "worked_examples": "prove a squared norm is convex; calculate the full step-size interval for a two-dimensional positive-definite quadratic; show a saddle with zero gradient; compare an exact quadratic Newton step with gradient descent",
      "labs": [
        {
          "id": "A",
          "description": "implements gradient descent on an ellipse"
        },
        {
          "id": "B",
          "description": "compares feature scaling and step-size choices"
        },
        {
          "id": "C",
          "description": "repairs divergence, a singular Hessian, and an inappropriate stopping test"
        }
      ],
      "exit_assessment": "derive the quadratic error recurrence, justify an admissible step size, and distinguish a theorem for convex objectives from a heuristic for neural networks",
      "visual": "contour paths with curvature and learning-rate controls",
      "connections": "optimisation, parameter fitting, and training diagnostics",
      "misconception": "treating a zero gradient as a certificate of a global optimum",
      "reading_keys": [
        "R7"
      ],
      "reading_description": "R7 convex functions and unconstrained minimisation"
    },
    {
      "number": 20,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Constrained optimisation Lagrange multipliers and duality",
        "zh-CN": "约束优化 拉格朗日乘子与对偶性"
      },
      "hours": 12,
      "prerequisites": [
        19
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "19",
      "part": "E",
      "outcomes": [
        "formulate constraints precisely",
        "derive a Lagrangian",
        "interpret multipliers",
        "write KKT conditions",
        "distinguish necessity, sufficiency, weak duality, and strong duality under their assumptions"
      ],
      "lesson_sequence": [
        "Equality/inequality constraints, feasible sets, active constraints, and projection.",
        "Equality-constrained extrema and Lagrange multipliers, including constraint-qualification caveats.",
        "Inequality constraints and KKT stationarity, primal/dual feasibility, and complementary slackness.",
        "Lagrange dual functions, weak duality, and dual bounds.",
        "Convex problems, Slater's condition in its appropriate form, and strong duality.",
        "Projected gradients, penalties versus exact constraints, and a small quadratic programme."
      ],
      "worked_examples": "minimise a quadratic subject to a line constraint; allocate a non-negative resource budget; solve a bound-constrained scalar problem; interpret an active multiplier as local sensitivity of the optimal value",
      "labs": [
        {
          "id": "A",
          "description": "projects points onto a box and simplex"
        },
        {
          "id": "B",
          "description": "implements projected gradient descent on a small convex problem"
        },
        {
          "id": "C",
          "description": "checks KKT residuals and diagnoses an infeasible result or an invalid strong-duality claim"
        }
      ],
      "exit_assessment": "solve one equality-constrained and one inequality-constrained example with all KKT conditions, then state the assumptions making the solution globally optimal",
      "visual": "objective contours with feasible region and active constraints",
      "connections": "scheduling, constrained learning, support vector machines, and later reinforcement-learning constraints",
      "misconception": "treating KKT conditions as an unconditional global certificate",
      "reading_keys": [
        "R7"
      ],
      "reading_description": "R7 duality and constrained optimisation"
    },
    {
      "number": 21,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Probability models conditioning and Bayes rule",
        "zh-CN": "概率模型 条件概率与贝叶斯法则"
      },
      "hours": 10,
      "prerequisites": [
        3,
        5
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "03 and 05",
      "part": "F",
      "outcomes": [
        "define a sample space and events",
        "apply probability axioms",
        "calculate conditional probabilities",
        "distinguish independence from disjointness",
        "explain a base-rate effect using Bayes' rule"
      ],
      "lesson_sequence": [
        "Probability models, events, axioms, complements, and countable additivity at an introductory level.",
        "Uniform versus non-uniform outcomes; union bounds and event inclusion–exclusion.",
        "Conditional probability, chain rules, and tree/table representations.",
        "Total probability and Bayes' rule; likelihood ratios and odds.",
        "Pairwise/mutual independence and conditional independence with explicit examples.",
        "Base rates, selection effects, and distinguishing a model's assumptions from observed frequencies."
      ],
      "worked_examples": "compute a posterior with prevalence 1%, sensitivity 90%, and false-positive rate 5%, obtaining `0.009 / 0.0585 = 2/13 ≈ 15.38%`; compare independent and disjoint events; construct three pairwise-independent events that are not mutually independent",
      "labs": [
        {
          "id": "A",
          "description": "enumerates exact probabilities in a finite weighted sample space"
        },
        {
          "id": "B",
          "description": "simulates the base-rate example at increasing sample sizes"
        },
        {
          "id": "C",
          "description": "diagnoses a reversed conditional probability and a selection-biased sample"
        }
      ],
      "exit_assessment": "specify a complete finite probability model, solve a conditional-probability tree, and explain each Bayes factor in words",
      "visual": "base-rate table with population counts",
      "connections": "reliability, false alarms, probabilistic classifiers, and diagnostic reasoning",
      "misconception": "assuming `P(A|B) = P(B|A)`",
      "reading_keys": [
        "R1",
        "R5"
      ],
      "reading_description": "R5 probability models and conditioning; R1 discrete probability"
    },
    {
      "number": 22,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Random variables distributions and transformations",
        "zh-CN": "随机变量 分布与变换"
      },
      "hours": 10,
      "prerequisites": [
        21
      ],
      "conditional_prerequisites": [
        {
          "module": 17,
          "branch": "continuous_probability"
        }
      ],
      "prerequisite_description": "21; 17 for the continuous branch",
      "part": "F",
      "outcomes": [
        "distinguish a random variable from an outcome or observation",
        "use PMFs, PDFs, and CDFs correctly",
        "select a distribution with stated assumptions",
        "compute probabilities after simple transformations"
      ],
      "lesson_sequence": [
        "Random variables as functions on a sample space; support and distribution.",
        "Discrete PMFs and CDFs; Bernoulli, categorical, binomial, geometric, and Poisson models.",
        "Continuous densities/CDFs; uniform, exponential, Gaussian, and Laplace models, requiring Module 17.",
        "Quantiles, inverse-CDF sampling, and the distinction between mass and density.",
        "Transformations of discrete variables; continuous monotone change of variables with its Jacobian factor.",
        "Model selection from the data-generating story; parameter conventions and support checks."
      ],
      "worked_examples": "calculate a binomial tail for a small trial count; derive a geometric waiting-time PMF; transform a uniform variable; explain why a continuous point has probability zero even when its density is positive",
      "labs": [
        {
          "id": "A",
          "description": "computes discrete tables and verifies normalisation"
        },
        {
          "id": "B",
          "description": "samples from selected distributions and compares empirical CDFs"
        },
        {
          "id": "C",
          "description": "repairs an omitted Jacobian and a distribution used with the wrong support; the CS branch uses a discrete transformation instead"
        }
      ],
      "exit_assessment": "derive one PMF from an experiment, calculate an event via a CDF, and justify the assumptions of a chosen distribution",
      "visual": "PMF/PDF/CDF and sampling explorer",
      "connections": "randomised algorithms, noise models, and classification likelihoods",
      "misconception": "interpreting a likelihood value for a continuous observation as its event probability",
      "reading_keys": [
        "R5",
        "R6"
      ],
      "reading_description": "R5 random variables; R6 probability and distributions"
    },
    {
      "number": 23,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Expectation joint distributions covariance and dependence",
        "zh-CN": "期望 联合分布 协方差与依赖关系"
      },
      "hours": 10,
      "prerequisites": [
        22
      ],
      "conditional_prerequisites": [
        {
          "module": 17,
          "branch": "continuous_probability"
        },
        {
          "module": 12,
          "branch": "vector_covariance_extension"
        }
      ],
      "prerequisite_description": "22; 17 for continuous integrals; 12 for the vector covariance extension",
      "part": "F",
      "outcomes": [
        "compute moments and conditional expectations",
        "use linearity without assuming independence",
        "distinguish uncorrelated from independent",
        "apply total expectation/variance",
        "interpret a covariance matrix when following the AI branch"
      ],
      "lesson_sequence": [
        "Expectation as a weighted sum or integral; expectation of a function.",
        "Variance, standard deviation, moments, and existence conditions.",
        "Joint/marginal distributions, dependence, and factorisation under independence.",
        "Covariance/correlation, variance of sums, and examples of dependence with zero covariance.",
        "Conditional expectation and the laws of total expectation and total variance.",
        "AI branch: covariance matrices, positive semidefiniteness, multivariate Gaussians, and the probability interpretation of PCA."
      ],
      "worked_examples": "compute a joint table's marginals; prove expectation is linear; use `Y = X²` with symmetric `X` to demonstrate dependence with zero covariance; calculate an expected search cost under a specified position distribution",
      "labs": [
        {
          "id": "A",
          "description": "computes exact moments from a joint discrete table"
        },
        {
          "id": "B",
          "description": "simulates dependent variables and compares covariance with scatter plots"
        },
        {
          "id": "C",
          "description": "demonstrates why an independence-based variance formula fails; the AI extension compares covariance eigenvectors with PCA directions"
        }
      ],
      "exit_assessment": "derive a variance formula with covariance terms and solve a conditional-expectation problem",
      "visual": "joint table linked to marginal plots",
      "connections": "average algorithm cost, uncertainty propagation, feature dependence, and model noise",
      "misconception": "equating zero correlation with independence outside special cases",
      "reading_keys": [
        "R5",
        "R6"
      ],
      "reading_description": "R5 expectation, joint variables, and conditioning; R6 distributions"
    },
    {
      "number": 24,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Limit theorems concentration and Monte Carlo",
        "zh-CN": "极限定理 集中不等式与蒙特卡洛方法"
      },
      "hours": 10,
      "prerequisites": [
        4,
        23
      ],
      "conditional_prerequisites": [
        {
          "module": 17,
          "branch": "continuous_probability"
        }
      ],
      "prerequisite_description": "04 and 23; 17 for continuous integration examples",
      "part": "F",
      "outcomes": [
        "state useful forms of the law of large numbers and central limit theorem",
        "calculate a Monte Carlo standard error",
        "apply a basic concentration bound",
        "distinguish statistical error from deterministic numerical error"
      ],
      "lesson_sequence": [
        "Sampling assumptions, independence, identical distribution, and what changes under dependence.",
        "Markov/Chebyshev inequalities and a complete elementary proof of Chebyshev.",
        "Weak law of large numbers through variance of an average under finite-variance assumptions.",
        "A standard i.i.d. finite-variance central limit theorem, standardisation, and limits of the approximation.",
        "Hoeffding's inequality for bounded independent variables; sample-size calculations and union bounds.",
        "Monte Carlo estimation, standard error, seed management, and variance reduction by paired/common random numbers."
      ],
      "worked_examples": "bound a Bernoulli sample mean's deviation; calculate how quadrupling samples changes standard error; compare CLT behaviour for a fair coin and a highly skewed finite-variance variable; expose a dependent sample that gives misleading error bars",
      "labs": [
        {
          "id": "A",
          "description": "compares repeated sample means with predicted variation"
        },
        {
          "id": "B",
          "description": "estimates a finite-event probability, with an optional continuous integral"
        },
        {
          "id": "C",
          "description": "diagnoses a copied seed, dependent draws, or a falsely precise report"
        }
      ],
      "exit_assessment": "derive the variance of an independent sample mean, compute a valid concentration bound, and report an estimate with its assumptions and uncertainty",
      "visual": "sampling-distribution explorer",
      "connections": "randomised algorithms, benchmark uncertainty, and stochastic training experiments",
      "misconception": "believing the CLT says the original observations become Gaussian",
      "reading_keys": [
        "R1",
        "R5"
      ],
      "reading_description": "R5 laws of large numbers and inference; R1 concentration topics"
    },
    {
      "number": 25,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Estimation likelihood and Bayesian updating",
        "zh-CN": "参数估计 似然与贝叶斯更新"
      },
      "hours": 12,
      "prerequisites": [
        16,
        17,
        23,
        24
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "16, 17, 23, and 24",
      "part": "F",
      "outcomes": [
        "distinguish a parameter from an estimator and estimate",
        "derive a maximum-likelihood estimator",
        "compare bias, variance, and mean squared error",
        "perform a conjugate Bayesian update",
        "explain the assumptions behind MAP regularisation"
      ],
      "lesson_sequence": [
        "Statistical models, parameters, samples, estimators, and identifiability.",
        "Likelihood versus probability as a function of different arguments; log-likelihood and independent sample factorisation.",
        "Maximum likelihood for Bernoulli and Gaussian models; boundaries and finite-sample issues.",
        "Bias, consistency, variance, mean squared error, and the sample-variance denominator.",
        "Bayesian updating with a beta–Bernoulli model; posterior mean, predictive probability, and credible intervals.",
        "MAP estimates, priors, Gaussian/Laplace noise, and the route from a probabilistic model to a loss or penalty."
      ],
      "worked_examples": "derive the Bernoulli sample-proportion MLE; derive a Gaussian mean estimate; update a `Beta(2, 2)` prior after eight successes and two failures; contrast posterior mode, mean, and a boundary MLE",
      "labs": [
        {
          "id": "A",
          "description": "computes likelihood curves and estimates"
        },
        {
          "id": "B",
          "description": "simulates estimator bias/variance over repeated datasets"
        },
        {
          "id": "C",
          "description": "repairs underflow, a missing normalising factor where it matters, or a prior mistakenly treated as data"
        }
      ],
      "exit_assessment": "derive two estimators, explain their model assumptions, and separate frequentist sampling uncertainty from posterior uncertainty",
      "visual": "likelihood and posterior explorer",
      "connections": "parameter fitting, loss construction, and regularisation",
      "misconception": "treating a likelihood as a normalised distribution over parameters without specifying a prior",
      "reading_keys": [
        "R5",
        "R6",
        "R8"
      ],
      "reading_description": "R5 inference; R6 models meeting data and regression; R8 introductory estimation"
    },
    {
      "number": 26,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Statistical inference experiments and regression",
        "zh-CN": "统计推断 实验设计与回归"
      },
      "hours": 12,
      "prerequisites": [
        12,
        19,
        25
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "12, 19, and 25",
      "part": "F",
      "outcomes": [
        "interpret confidence intervals and p-values correctly",
        "design a comparison with an appropriate unit of analysis",
        "derive basic regression fits from likelihoods",
        "distinguish fit, prediction, and causal interpretation",
        "diagnose leakage and repeated testing"
      ],
      "lesson_sequence": [
        "Sampling distributions, standard errors, confidence intervals, and coverage by repeated sampling.",
        "Null hypotheses, test statistics, significance, power, effect sizes, and p-value interpretation.",
        "Randomisation, pairing, confounding, sampling units, and limits of observational conclusions.",
        "Bootstrap and permutation tests with exchangeability and independence assumptions.",
        "Linear/logistic regression as probabilistic models; residual checks and uncertainty conditions.",
        "Train/validation/test separation, cross-validation, multiple comparisons, and selecting models without using test outcomes."
      ],
      "worked_examples": "simulate nominal interval coverage; compare paired and unpaired model results; derive logistic negative log-likelihood; show a large-sample tiny effect with a small p-value; identify preprocessing leakage",
      "labs": [
        {
          "id": "A",
          "description": "implements a bootstrap interval and permutation comparison"
        },
        {
          "id": "B",
          "description": "fits a small regression/classification model and inspects residuals or calibration"
        },
        {
          "id": "C",
          "description": "repairs test-set tuning and a benchmark that treats correlated observations as independent"
        }
      ],
      "exit_assessment": "write an experimental protocol before seeing results, report an effect and uncertainty, and explain why an interval or p-value does not establish a causal mechanism",
      "visual": "coverage and resampling explorer",
      "connections": "model evaluation, A/B comparisons, and reproducible empirical claims",
      "misconception": "interpreting a p-value as the probability the null hypothesis is true",
      "reading_keys": [
        "R8"
      ],
      "reading_description": "R8 regression, resampling, and multiple testing"
    },
    {
      "number": 27,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Information theory entropy and probabilistic objectives",
        "zh-CN": "信息论 熵与概率目标函数"
      },
      "hours": 10,
      "prerequisites": [
        16,
        23,
        25
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "16, 23, and 25",
      "part": "G",
      "outcomes": [
        "compute discrete entropy and cross-entropy",
        "explain KL divergence's direction and support conditions",
        "derive cross-entropy from likelihood",
        "interpret mutual information",
        "calculate perplexity with a stated log base and token unit"
      ],
      "lesson_sequence": [
        "Surprisal, coding intuition, logarithm bases, and units of bits/nats.",
        "Discrete entropy, conditional entropy, and chain rules.",
        "Cross-entropy and its relationship to expected code length and negative log-likelihood.",
        "KL divergence, non-negativity using a stated inequality, asymmetry, and support mismatches.",
        "Mutual information and dependence; data-processing intuition with a stated theorem rather than an omitted proof.",
        "Perplexity, sequence factorisation, and a preview of differential entropy and its different properties."
      ],
      "worked_examples": "compare entropy of a fair and biased coin; calculate cross-entropy for two categorical distributions; obtain infinite KL when a required probability is zero; convert a mean token loss in nats into perplexity",
      "labs": [
        {
          "id": "A",
          "description": "computes entropy/KL with careful zero handling"
        },
        {
          "id": "B",
          "description": "compares a small frequency-based language model with a uniform baseline"
        },
        {
          "id": "C",
          "description": "diagnoses mixed log bases, invalid probabilities, and meaningless perplexity comparisons across tokenisations"
        }
      ],
      "exit_assessment": "derive categorical cross-entropy from independent observations, prove a discrete KL non-negativity statement with its conditions, and interpret perplexity without claiming it measures all aspects of quality",
      "visual": "categorical-distribution and coding explorer",
      "connections": "compression, classification losses, and language-model objectives",
      "misconception": "describing KL divergence as a symmetric metric",
      "reading_keys": [
        "R6",
        "R9"
      ],
      "reading_description": "R9 entropy, relative entropy, and mutual information; R6 likelihood-based classification"
    },
    {
      "number": 28,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Stochastic optimisation regularisation and training dynamics",
        "zh-CN": "随机优化 正则化与训练动态"
      },
      "hours": 12,
      "prerequisites": [
        19,
        23,
        25
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "19, 23, and 25",
      "part": "G",
      "outcomes": [
        "derive a mini-batch gradient estimator",
        "state when it is unbiased",
        "explain batch variance and sampling assumptions",
        "distinguish regularisation methods",
        "diagnose noisy, ill-conditioned, and unstable training"
      ],
      "lesson_sequence": [
        "Expected risk versus empirical risk, finite dataset objectives, and stochastic gradient estimates.",
        "Sampling with/without replacement, batch size, gradient variance, and correlation.",
        "SGD, step schedules, momentum, and explicitly defined adaptive updates as methods to inspect rather than unconditional guarantees.",
        "Ridge/lasso geometry, MAP connections, and non-smooth penalties/subgradient conventions.",
        "Early stopping, explicit penalties, and when weight decay coincides with a penalty gradient.",
        "Conditioning, clipping, validation curves, stochastic stopping, and seed-to-seed variability."
      ],
      "worked_examples": "derive `E[g_batch] = ∇L` for uniform sampling; calculate gradient variance for a small dataset; compare ridge coefficients with unregularised ones; show why Adam-style decoupled weight decay differs from adding an L2 gradient",
      "labs": [
        {
          "id": "A",
          "description": "compares full-batch, single-sample, and mini-batch updates"
        },
        {
          "id": "B",
          "description": "studies regularisation and feature scaling on a fixed split"
        },
        {
          "id": "C",
          "description": "diagnoses a bad learning rate, correlated batches, and a test-based stopping rule"
        }
      ],
      "exit_assessment": "identify the source of randomness, derive an unbiased gradient estimate under explicit assumptions, and report training/evaluation variation across seeds",
      "visual": "noisy gradient paths and regularisation slider",
      "connections": "neural-network training and reliable optimiser comparisons",
      "misconception": "interpreting lower training loss as proof of better generalisation",
      "reading_keys": [
        "R6",
        "R10"
      ],
      "reading_description": "R6 optimisation and regression; R10 stochastic gradient descent. Verify any named optimiser's primary paper when writing its algorithm box"
    },
    {
      "number": 29,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Numerical computation conditioning and reliable experiments",
        "zh-CN": "数值计算 条件数与可靠实验"
      },
      "hours": 12,
      "prerequisites": [
        14,
        18,
        19
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "14, 18, and 19",
      "part": "G",
      "outcomes": [
        "distinguish conditioning from algorithmic stability",
        "explain rounding/cancellation",
        "implement stable probability calculations",
        "choose a suitable linear solver",
        "report approximation error and reproducibility limits"
      ],
      "lesson_sequence": [
        "Floating-point representation, rounding, machine epsilon, overflow/underflow, and representable spacing.",
        "Absolute/relative error, catastrophic cancellation, stable summation, and safe comparisons.",
        "Condition numbers, perturbation sensitivity, forward/backward error, and stable algorithms.",
        "Linear solves, QR/SVD, iterative methods, and why normal equations square the two-norm condition number for full-column-rank matrices.",
        "Stable sigmoid, softmax, log-sum-exp, log probabilities, and gradient checking away from kinks.",
        "Precision versus memory/cost, operation counts, random seeds, tolerance-based regression checks, and reproducible reports."
      ],
      "worked_examples": "explain the representation error in `0.1 + 0.2`; compare two algebraically equivalent expressions near cancellation; stabilise logits `[1000, 1001, 1002]`; perturb an ill-conditioned system and distinguish a small residual from small solution error",
      "labs": [
        {
          "id": "A",
          "description": "measures rounding/summation errors at several precisions"
        },
        {
          "id": "B",
          "description": "compares solves and condition estimates"
        },
        {
          "id": "C",
          "description": "repairs overflowing cross-entropy and a misleading finite-difference gradient check"
        }
      ],
      "exit_assessment": "provide a numerical error diagnosis with a proposed repair, compute and interpret a condition estimate, and justify tolerances from scale and precision",
      "visual": "precision and conditioning explorer",
      "connections": "robust scientific code, mixed-precision awareness, and AI serving calculations",
      "misconception": "confusing an accurate computation of an ill-conditioned problem with an insensitive solution",
      "reading_keys": [
        "R2",
        "R11",
        "R12"
      ],
      "reading_description": "R11 and R12 numerical linear algebra documentation; R2 decomposition lectures. Check documentation against the pinned lab environment when writing examples"
    },
    {
      "number": 30,
      "status": "available",
      "kind": "taught",
      "title": {
        "en": "Generalisation kernels and mathematical learning theory",
        "zh-CN": "泛化 核方法与数学学习理论"
      },
      "hours": 12,
      "prerequisites": [
        12,
        24,
        26,
        27
      ],
      "conditional_prerequisites": [
        {
          "module": 20,
          "branch": "svm_duality_extension"
        }
      ],
      "prerequisite_description": "12, 24, 26, and 27; 20 for the full SVM duality extension",
      "part": "G",
      "outcomes": [
        "distinguish empirical from population risk",
        "explain why repeated model selection affects evaluation",
        "derive a finite-class uniform bound under assumptions",
        "verify a small positive-semidefinite kernel matrix",
        "connect capacity and regularisation without overclaiming"
      ],
      "lesson_sequence": [
        "Hypothesis classes, empirical/population risk, and the i.i.d. sampling assumption.",
        "Fixed-model concentration versus a uniform statement over a finite class; use a union bound and Hoeffding.",
        "Bias–variance decomposition for squared loss with the randomness and target specified.",
        "Capacity, VC-dimension intuition, distribution shift, and the limits of introductory bounds for modern networks.",
        "Feature maps, kernels, Gram matrices, positive semidefiniteness, and kernel ridge regression.",
        "Margins and an introductory SVM formulation; constrained-dual derivation is an optional Module 20 extension."
      ],
      "worked_examples": "bound risk error for a finite set of candidate classifiers; derive a kernel from a two-dimensional feature map; show that a symmetric matrix need not be a valid kernel matrix; calculate an overfit polynomial example",
      "labs": [
        {
          "id": "A",
          "description": "measures optimism from selecting among many models"
        },
        {
          "id": "B",
          "description": "implements kernel ridge regression with a solve"
        },
        {
          "id": "C",
          "description": "diagnoses an indefinite similarity matrix and a failure under distribution shift"
        }
      ],
      "exit_assessment": "derive a finite-class risk bound with explicit sample-size, independence, loss-range, and confidence assumptions; prove a feature-map Gram matrix is PSD; explain what the result does and does not predict about a neural network",
      "visual": "capacity, split, and kernel explorer",
      "connections": "model selection, kernel methods, and critical reading of learning claims",
      "misconception": "treating a loose sufficient bound as an exact practical sample-size requirement",
      "reading_keys": [
        "R6",
        "R8",
        "R10"
      ],
      "reading_description": "R10 uniform convergence and kernels; R6 support vector machines; R8 model assessment"
    },
    {
      "number": 31,
      "status": "available",
      "kind": "capstone",
      "title": {
        "en": "CS capstone verified dependency planner",
        "zh-CN": "计算机科学综合项目 可验证的依赖规划器"
      },
      "hours": 14,
      "prerequisites": [
        4,
        6,
        7,
        8,
        21,
        24
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "04, 06, 07, 08, 21, and 24",
      "part": "H",
      "problem": "build a small planner for tasks with prerequisites and finite random duration models. It must either return a valid dependency order and predicted completion times or provide a cycle witness. Use a DAG model with unlimited parallel workers for critical-path calculations; do not imply this solves a resource-constrained scheduling problem",
      "stages": [
        "**Specify and model, 2 hours:** use 8–12 tasks, including a diamond dependency, an isolated task, and an invalid cycle case. Define task IDs, positive finite durations, invalid-input behaviour, and whether edges represent immediate prerequisites.",
        "**Derive and prove, 4 hours:** prove the returned topological order satisfies every edge; derive earliest-finish recurrence `F(v) = duration(v) + max F(u)` over predecessors, with empty maximum zero; prove the recurrence by induction over the order; count operations as `O(V + E)` for the relevant passes.",
        "**Implement and challenge, 4 hours:** implement cycle detection, topological order, and critical-path times; add a simple modular checksum as an educational consistency check; include duplicate IDs, missing prerequisites, zero-task input, and deliberately corrupted data.",
        "**Quantify and report, 4 hours:** assign small discrete duration distributions to a 3–4 task subgraph; enumerate exact completion-time outcomes; simulate the same model and report Monte Carlo error; compare independent draws with a dependent shared-delay case."
      ],
      "exit_assessment": "defend the proof obligations, explain why `E[max(X, Y)]` generally differs from `max(E[X], E[Y])`, and separate deterministic operation counts from random task duration. Demonstrate a cycle witness and a counterexample to a checksum guaranteeing integrity",
      "rubric": "modelling/contract 20%; correctness and termination arguments 25%; implementation and fault diagnosis 20%; complexity and uncertainty analysis 20%; communication and reproducibility 15%. Recommend 80% overall, with no missing proof or uncertainty explanation",
      "visual": "dependency graph with the critical path and completion-time distribution",
      "optional_extension": "add a fixed worker limit and explain why the previous recurrence ceases to define the full scheduling problem; formulate a constrained optimisation approach after Module 20",
      "deliverables": "a written input/output contract; graph model and assumptions; Python implementation; invariant/induction proofs; operation-count analysis; exact and simulated uncertainty calculations; and a concise report explaining limitations"
    },
    {
      "number": 32,
      "status": "available",
      "kind": "capstone",
      "title": {
        "en": "AI capstone a learning pipeline with mathematical checks",
        "zh-CN": "人工智能综合项目 具备数学检验的学习流程"
      },
      "hours": 16,
      "prerequisites": [
        14,
        18,
        26,
        27,
        28,
        29,
        30
      ],
      "conditional_prerequisites": [],
      "prerequisite_description": "14, 18, 26, 27, 28, 29, and 30",
      "part": "H",
      "problem": "implement and explain a small binary classifier from data generation to an honest final evaluation. Include ill-scaled and nearly dependent features so the mathematics predicts failures that the learner must then repair. No GPU, paid API, or external dataset is required",
      "stages": [
        "**Define and inspect, 4 hours:** state the generative model and assumptions; split before fitting transformations; fit scaling only on training data; inspect singular values; derive Bernoulli negative log-likelihood and a stated L2 penalty convention.",
        "**Derive and implement, 4 hours:** implement stable sigmoid/log-loss and a NumPy logistic model; derive the gradient with dimensions and batch normalisation; verify derivatives numerically on smooth small inputs; compare an unscaled unstable fit with a repaired fit.",
        "**Select and diagnose, 4 hours:** compare learning rates and penalties using validation data only; include a constant class-probability baseline; test four required faults: preprocessing leakage, overflowing logits, a missing gradient factor, and a misleading metric under changed class prevalence.",
        "**Evaluate and explain, 4 hours:** fix choices before final test evaluation; report log-loss, a chosen threshold metric, and calibration evidence with stated uncertainty assumptions; explain seed variation; provide one complete mathematical trace of a prediction and update."
      ],
      "exit_assessment": "a learner must explain a mismatch between mathematical and numerical gradients, why training data define preprocessing parameters, how regularisation changes the objective, and why test performance is an estimate rather than a general guarantee. Results must include actual generated outputs when the capstone is implemented",
      "rubric": "model/assumptions 15%; derivations and gradient verification 25%; numerical implementation 20%; experimental design and evaluation 25%; reproducibility and explanation 15%. Recommend 80% overall, with all leakage and derivative faults correctly diagnosed",
      "visual": "pipeline diagram, training curves, calibration plot, and singular-value plot",
      "optional_extension": "build a two-layer network or a three-token attention calculation; derive its backward pass and contrast its numerical behaviour with the linear model. Keep this outside the sixteen-hour core",
      "dataset_specification": "generate 1,000 independent examples with seed 7. Draw independent standard-normal latent variables `z1, z2, z3, e1, e2`; set features to `(z1, 100z2, z1 + 0.02e1, z3, e2)`; draw the label from a Bernoulli distribution with success probability `sigmoid(1.5z1 − 2z2 + 0.5z3)`. Use a separate split seed 11 to allocate disjoint 60%/20%/20% training/validation/test sets, preserving class proportions as closely as integer counts permit. Record the random generator and package versions",
      "mathematical_appendix": "equations for the generative model, likelihood, objective, and gradient; shapes for every array; a short explanation of Hessian curvature and feature scaling; a rank/PCA interpretation; and an empirical-versus-population-risk statement with limits. Fit a small training-only PCA variant as a comparison, keeping component selection on validation data"
    }
  ],
  "reading_sources": [
    {
      "key": "R1",
      "title": "MIT Mathematics for Computer Science",
      "url": "https://ocw.mit.edu/courses/6-042j-mathematics-for-computer-science-spring-2015/",
      "proposed_use": "Discrete definitions, proofs, structures, counting, and probability in 01–08 and 21–24"
    },
    {
      "key": "R2",
      "title": "MIT Linear Algebra",
      "url": "https://ocw.mit.edu/courses/18-06-linear-algebra-spring-2010/",
      "proposed_use": "Systems, spaces, projections, eigenvalues, and decompositions in 09–14"
    },
    {
      "key": "R3",
      "title": "MIT Single Variable Calculus",
      "url": "https://ocw.mit.edu/courses/18-01sc-single-variable-calculus-fall-2010/",
      "proposed_use": "Derivatives, integration, approximation, and infinite processes in 15–17"
    },
    {
      "key": "R4",
      "title": "MIT Multivariable Calculus",
      "url": "https://ocw.mit.edu/courses/18-02sc-multivariable-calculus-fall-2010/",
      "proposed_use": "Partial derivatives, chain rules, and multivariable geometry in 18"
    },
    {
      "key": "R5",
      "title": "MIT Probabilistic Systems Analysis and Applied Probability",
      "url": "https://ocw.mit.edu/courses/6-041sc-probabilistic-systems-analysis-and-applied-probability-fall-2013/",
      "proposed_use": "Probability, random variables, expectation, limit theorems, and inference in 21–25"
    },
    {
      "key": "R6",
      "title": "Mathematics for Machine Learning by Deisenroth Faisal and Ong",
      "url": "https://mml-book.com/",
      "proposed_use": "Connections among linear algebra, vector calculus, probability, optimisation, regression, PCA, and SVMs"
    },
    {
      "key": "R7",
      "title": "Convex Optimization by Boyd and Vandenberghe",
      "url": "https://web.stanford.edu/~boyd/cvxbook/",
      "proposed_use": "Convexity, duality, and optimisation in 19–20"
    },
    {
      "key": "R8",
      "title": "An Introduction to Statistical Learning",
      "url": "https://www.statlearning.com/",
      "proposed_use": "Regression, resampling, model assessment, and statistical practice in 25–26 and 30"
    },
    {
      "key": "R9",
      "title": "Stanford EE376A Information Theory course outline",
      "url": "https://web.stanford.edu/class/ee376a/outline.html",
      "proposed_use": "Entropy, relative entropy, mutual information, and coding foundations in 27"
    },
    {
      "key": "R10",
      "title": "Understanding Machine Learning by Shalev Shwartz and Ben David",
      "url": "https://www.cambridge.org/core/books/understanding-machine-learning/3059695661405D25673058E43C8BE2A6",
      "proposed_use": "Uniform convergence, model selection, SGD, and kernels in 28 and 30; publisher page with chapter catalogue"
    },
    {
      "key": "R11",
      "title": "NumPy linear algebra reference",
      "url": "https://numpy.org/doc/stable/reference/routines.linalg.html",
      "proposed_use": "API and shape checks for linear algebra labs, especially 09–14 and 29"
    },
    {
      "key": "R12",
      "title": "SciPy linear algebra tutorial",
      "url": "https://docs.scipy.org/doc/scipy/tutorial/linalg.html",
      "proposed_use": "Reference solver behaviour and implementation comparisons in 12 and 29"
    }
  ],
  "published_modules": [
    1,
    2,
    3,
    4,
    5,
    6,
    7,
    8,
    9,
    10,
    11,
    12,
    13,
    14,
    15,
    16,
    17,
    18,
    19,
    20,
    21,
    22,
    23,
    24,
    25,
    26,
    27,
    28,
    29,
    30,
    31,
    32
  ]
}
