Module 9 · Dependence, Regression, and Model Foundations Lesson 90 of 120
Regression Assumptions, Heteroskedasticity, and Multicollinearity
Diagnosing unstable uncertainty and redundant predictors.
Transcript
19 sentences · select one to jump thereCheck your understanding
Does a robust standard error automatically solve omitted-variable confounding?
Code lab
Run it yourself
The lesson source in 7 languages. Edit it, run TypeScript and Python right here, and compare with the expected output.
/**
* Fintech Math Bootcamp · Lesson 090 of 120
* Regression Assumptions, Heteroskedasticity, and Multicollinearity
* Module 09: Dependence, Regression, and Model Foundations
*
* Scenario: Diagnosing unstable uncertainty and redundant predictors
* Rule: nonconstant error variance; collinearity creates unstable coefficient identification
*
* Try it: Does a robust standard error automatically solve omitted-variable confounding?
*
* Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
* Free course: https://courses.thefintechbuilder.com
* Synthetic teaching example, not financial advice or a production library.
*/
export function lesson090() {
const lowErrors=[-1,1],highErrors=[-4,4];
const meanSquare=(a:number[])=>a.reduce((s,x)=>s+x*x,0)/a.length;
const x=[1,2,3,4],duplicate=x.map(v=>2*v);
const avg=(a:number[])=>a.reduce((s,v)=>s+v,0)/a.length;
const dx=x.map(v=>v-avg(x)),dd=duplicate.map(v=>v-avg(duplicate));
const dot=(a:number[],b:number[])=>a.reduce((s,v,i)=>s+v*b[i],0);
const r2Between=dot(dx,dd)**2/(dot(dx,dx)*dot(dd,dd)); // 1 means VIF is infinite
const result={lowSpread:meanSquare(lowErrors),highSpread:meanSquare(highErrors),
exactRedundancy:r2Between===1};
return result;
}
export const checkedResult = {"lowSpread":1,"highSpread":16,"exactRedundancy":true};
// Run this file directly: npx tsx lessons/09-dependence-regression-and-model-foundations/090-regression-assumptions-heteroskedasticity-and-multicollinearity.ts
if (process.argv[1] && import.meta.url.endsWith(process.argv[1].replace(/\\/g, "/").split("/").pop()!)) {
console.log(JSON.stringify(lesson090(), null, 2));
}
Your output
Press Run to execute the code in your browser.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}"""
Fintech Math Bootcamp · Lesson 090 of 120
Regression Assumptions, Heteroskedasticity, and Multicollinearity
Module 09: Dependence, Regression, and Model Foundations
Scenario: Diagnosing unstable uncertainty and redundant predictors
Rule: nonconstant error variance; collinearity creates unstable coefficient identification
Try it: Does a robust standard error automatically solve omitted-variable confounding?
Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
Free course: https://courses.thefintechbuilder.com
Synthetic teaching example, not financial advice or a production library.
"""
import json
def mean_square(a):
total = 0
for x in a:
total += x * x
return total / len(a)
def average(a):
total = 0
for v in a:
total += v
return total / len(a)
def dot(a, b):
total = 0
for ai, bi in zip(a, b):
total += ai * bi
return total
def lesson_090():
low_errors, high_errors = [-1, 1], [-4, 4]
x = [1, 2, 3, 4]
duplicate = [2 * v for v in x]
dx = [v - average(x) for v in x]
dd = [v - average(duplicate) for v in duplicate]
r2_between = dot(dx, dd) ** 2 / (dot(dx, dx) * dot(dd, dd)) # 1 means VIF is infinite
return {
"lowSpread": mean_square(low_errors),
"highSpread": mean_square(high_errors),
"exactRedundancy": r2_between == 1,
}
if __name__ == "__main__":
print(json.dumps(lesson_090(), indent=2))
Your output
Press Run to execute the code in your browser.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}// Fintech Math Bootcamp - Lesson 090 of 120
// Regression Assumptions, Heteroskedasticity, and Multicollinearity
// Module 09: Dependence, Regression, and Model Foundations
//
// Scenario: Diagnosing unstable uncertainty and redundant predictors
// Rule: nonconstant error variance; collinearity creates unstable coefficient identification
//
// Try it: Does a robust standard error automatically solve omitted-variable confounding?
//
// Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
// Free course: https://courses.thefintechbuilder.com
// Synthetic teaching example, not financial advice or a production library.
import java.util.ArrayList;
import java.util.Arrays;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
public class Main {
static double meanSquare(double[] a) {
double total = 0;
for (double x : a) total += x * x;
return total / a.length;
}
static double[] deviations(double[] values) {
double total = 0;
for (double v : values) total += v;
double m = total / values.length;
double[] out = new double[values.length];
for (int i = 0; i < values.length; i++) out[i] = values[i] - m;
return out;
}
static double dot(double[] a, double[] b) {
double total = 0;
for (int i = 0; i < a.length; i++) total += a[i] * b[i];
return total;
}
static Map<String, Object> lesson090() {
double[] lowErrors = {-1, 1}, highErrors = {-4, 4};
double[] x = {1, 2, 3, 4};
double[] duplicate = new double[x.length];
for (int i = 0; i < x.length; i++) duplicate[i] = 2 * x[i];
double[] dx = deviations(x), dd = deviations(duplicate);
double r2Between = Math.pow(dot(dx, dd), 2) / (dot(dx, dx) * dot(dd, dd)); // 1 means VIF is infinite
Map<String, Object> result = new LinkedHashMap<String, Object>();
result.put("lowSpread", meanSquare(lowErrors));
result.put("highSpread", meanSquare(highErrors));
result.put("exactRedundancy", r2Between == 1);
return result;
}
public static void main(String[] args) {
System.out.println(toJson(lesson090(), ""));
}
// Minimal JSON writer: two-space indent, whole numbers without a decimal point, NaN as null.
static String toJson(Object value, String indent) {
if (value == null) return "null";
if (value instanceof Boolean) return value.toString();
if (value instanceof Number) return formatNumber(((Number) value).doubleValue());
if (value instanceof String) return quote((String) value);
if (value instanceof double[]) {
List<Object> boxed = new ArrayList<Object>();
for (double d : (double[]) value) boxed.add(d);
return toJson(boxed, indent);
}
if (value instanceof Object[]) return toJson(Arrays.asList((Object[]) value), indent);
String inner = indent + " ";
StringBuilder out = new StringBuilder();
if (value instanceof Map) {
Map<?, ?> map = (Map<?, ?>) value;
if (map.isEmpty()) return "{}";
out.append("{\n");
int i = 0;
for (Map.Entry<?, ?> entry : map.entrySet()) {
out.append(inner).append(quote(entry.getKey().toString())).append(": ")
.append(toJson(entry.getValue(), inner));
out.append(++i < map.size() ? ",\n" : "\n");
}
return out.append(indent).append("}").toString();
}
List<?> list = (List<?>) value;
if (list.isEmpty()) return "[]";
out.append("[\n");
for (int i = 0; i < list.size(); i++) {
out.append(inner).append(toJson(list.get(i), inner));
out.append(i + 1 < list.size() ? ",\n" : "\n");
}
return out.append(indent).append("]").toString();
}
static String formatNumber(double x) {
if (Double.isNaN(x) || Double.isInfinite(x)) return "null";
if (x == Math.rint(x) && Math.abs(x) < 1e15) return Long.toString((long) x);
return Double.toString(x);
}
static String quote(String s) {
StringBuilder out = new StringBuilder("\"");
for (char c : s.toCharArray()) {
if (c == '"' || c == '\\') out.append('\\').append(c);
else if (c == '\n') out.append("\\n");
else if (c < 0x20) out.append(String.format("\\u%04x", (int) c));
else out.append(c);
}
return out.append('"').toString();
}
}
No browser runner for Java yet
Read the code here, then run it in your own toolchain or a ready-made cloud workspace.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}// Fintech Math Bootcamp · Lesson 090 of 120
// Regression Assumptions, Heteroskedasticity, and Multicollinearity
// Module 09: Dependence, Regression, and Model Foundations
//
// Scenario: Diagnosing unstable uncertainty and redundant predictors
// Rule: nonconstant error variance; collinearity creates unstable coefficient identification
//
// Try it: Does a robust standard error automatically solve omitted-variable confounding?
//
// Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
// Free course: https://courses.thefintechbuilder.com
// Synthetic teaching example, not financial advice or a production library.
package main
import (
"encoding/json"
"fmt"
"math"
)
type Lesson090Result struct {
LowSpread float64 `json:"lowSpread"`
HighSpread float64 `json:"highSpread"`
ExactRedundancy bool `json:"exactRedundancy"`
}
func meanSquare(a []float64) float64 {
total := 0.0
for _, x := range a {
total += x * x
}
return total / float64(len(a))
}
func deviations(values []float64) []float64 {
total := 0.0
for _, v := range values {
total += v
}
m := total / float64(len(values))
out := make([]float64, len(values))
for i, v := range values {
out[i] = v - m
}
return out
}
func dot(a, b []float64) float64 {
total := 0.0
for i := range a {
total += a[i] * b[i]
}
return total
}
func lesson090() Lesson090Result {
lowErrors := []float64{-1, 1}
highErrors := []float64{-4, 4}
x := []float64{1, 2, 3, 4}
duplicate := make([]float64, len(x))
for i, v := range x {
duplicate[i] = 2 * v
}
dx, dd := deviations(x), deviations(duplicate)
r2Between := math.Pow(dot(dx, dd), 2) / (dot(dx, dx) * dot(dd, dd)) // 1 means VIF is infinite
return Lesson090Result{
LowSpread: meanSquare(lowErrors),
HighSpread: meanSquare(highErrors),
ExactRedundancy: r2Between == 1,
}
}
func main() {
out, err := json.MarshalIndent(lesson090(), "", " ")
if err != nil {
panic(err)
}
fmt.Println(string(out))
}
No browser runner for Go yet
Read the code here, then run it in your own toolchain or a ready-made cloud workspace.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}// Fintech Math Bootcamp · Lesson 090 of 120
// Regression Assumptions, Heteroskedasticity, and Multicollinearity
// Module 09: Dependence, Regression, and Model Foundations
//
// Scenario: Diagnosing unstable uncertainty and redundant predictors
// Rule: nonconstant error variance; collinearity creates unstable coefficient identification
//
// Try it: Does a robust standard error automatically solve omitted-variable confounding?
//
// Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
// Free course: https://courses.thefintechbuilder.com
// Synthetic teaching example, not financial advice or a production library.
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <iostream>
#include <optional>
#include <stdexcept>
#include <string>
#include <utility>
#include <vector>
// A minimal JSON value, enough to print this lesson's result.
struct Json {
enum class Kind { Null, Bool, Number, String, Array, Object };
Kind kind = Kind::Null;
bool flag = false;
double number = 0.0;
std::string text;
std::vector<std::string> keys; // object keys, parallel to items
std::vector<Json> items; // array elements or object values
Json() = default;
Json(bool value) : kind(Kind::Bool), flag(value) {}
Json(int value) : kind(Kind::Number), number(value) {}
Json(double value) : kind(Kind::Number), number(value) {}
Json(const char* value) : kind(Kind::String), text(value) {}
Json(const std::string& value) : kind(Kind::String), text(value) {}
Json(const std::vector<double>& values) : kind(Kind::Array) {
for (double v : values) items.push_back(Json(v));
}
};
Json jsonArray(const std::vector<Json>& values) {
Json array;
array.kind = Json::Kind::Array;
array.items = values;
return array;
}
Json jsonObject(const std::vector<std::pair<std::string, Json>>& fields) {
Json object;
object.kind = Json::Kind::Object;
for (const auto& field : fields) {
object.keys.push_back(field.first);
object.items.push_back(field.second);
}
return object;
}
// Shortest decimal form that reads back as the same double.
std::string formatNumber(double x) {
if (!std::isfinite(x)) return "null";
char buffer[32];
if (x == std::floor(x) && std::fabs(x) < 1e15) {
std::snprintf(buffer, sizeof buffer, "%.0f", x);
return buffer;
}
for (int precision = 1; precision <= 17; ++precision) {
std::snprintf(buffer, sizeof buffer, "%.*g", precision, x);
if (std::strtod(buffer, nullptr) == x) break;
}
return buffer;
}
std::string quote(const std::string& s) {
std::string out = "\"";
for (char c : s) {
if (c == '"' || c == '\\') { out += '\\'; out += c; }
else if (c == '\n') out += "\\n";
else out += c;
}
return out + "\"";
}
std::string toJson(const Json& value, const std::string& indent = "") {
switch (value.kind) {
case Json::Kind::Null: return "null";
case Json::Kind::Bool: return value.flag ? "true" : "false";
case Json::Kind::Number: return formatNumber(value.number);
case Json::Kind::String: return quote(value.text);
default: break;
}
const bool isObject = value.kind == Json::Kind::Object;
if (value.items.empty()) return isObject ? "{}" : "[]";
const std::string inner = indent + " ";
std::string out = isObject ? "{\n" : "[\n";
for (std::size_t i = 0; i < value.items.size(); ++i) {
out += inner;
if (isObject) out += quote(value.keys[i]) + ": ";
out += toJson(value.items[i], inner);
out += i + 1 < value.items.size() ? ",\n" : "\n";
}
return out + indent + (isObject ? "}" : "]");
}
double meanSquare(const std::vector<double>& a) {
double total = 0.0;
for (double x : a) total += x * x;
return total / a.size();
}
std::vector<double> deviations(const std::vector<double>& values) {
double total = 0.0;
for (double v : values) total += v;
const double m = total / values.size();
std::vector<double> out;
for (double v : values) out.push_back(v - m);
return out;
}
double dot(const std::vector<double>& a, const std::vector<double>& b) {
double total = 0.0;
for (std::size_t i = 0; i < a.size(); ++i) total += a[i] * b[i];
return total;
}
Json lesson090() {
const std::vector<double> lowErrors = {-1, 1}, highErrors = {-4, 4};
const std::vector<double> x = {1, 2, 3, 4};
std::vector<double> duplicate;
for (double v : x) duplicate.push_back(2 * v);
const std::vector<double> dx = deviations(x), dd = deviations(duplicate);
const double r2Between = std::pow(dot(dx, dd), 2) / (dot(dx, dx) * dot(dd, dd)); // 1 means VIF is infinite
return jsonObject({
{"lowSpread", meanSquare(lowErrors)},
{"highSpread", meanSquare(highErrors)},
{"exactRedundancy", r2Between == 1},
});
}
int main() {
std::cout << toJson(lesson090()) << '\n';
return 0;
}
No browser runner for C++ yet
Read the code here, then run it in your own toolchain or a ready-made cloud workspace.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}// Fintech Math Bootcamp · Lesson 090 of 120
// Regression Assumptions, Heteroskedasticity, and Multicollinearity
// Module 09: Dependence, Regression, and Model Foundations
//
// Scenario: Diagnosing unstable uncertainty and redundant predictors
// Rule: nonconstant error variance; collinearity creates unstable coefficient identification
//
// Try it: Does a robust standard error automatically solve omitted-variable confounding?
//
// Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
// Free course: https://courses.thefintechbuilder.com
// Synthetic teaching example, not financial advice or a production library.
/// A minimal JSON value, enough to print this lesson's result.
#[allow(dead_code)]
enum Json {
Null,
Bool(bool),
Num(f64),
Str(String),
Arr(Vec<Json>),
Obj(Vec<(String, Json)>),
}
#[allow(dead_code)]
impl Json {
fn obj(fields: Vec<(&str, Json)>) -> Json {
Json::Obj(fields.into_iter().map(|(k, v)| (k.to_string(), v)).collect())
}
fn nums(values: &[f64]) -> Json {
Json::Arr(values.iter().map(|&v| Json::Num(v)).collect())
}
/// Pretty-prints with two-space indentation.
fn pretty(&self, indent: &str) -> String {
let inner = format!("{} ", indent);
match self {
Json::Null => "null".to_string(),
Json::Bool(b) => b.to_string(),
Json::Num(x) => format_number(*x),
Json::Str(s) => quote(s),
Json::Arr(items) if items.is_empty() => "[]".to_string(),
Json::Obj(fields) if fields.is_empty() => "{}".to_string(),
Json::Arr(items) => {
let body: Vec<String> = items
.iter()
.map(|v| format!("{}{}", inner, v.pretty(&inner)))
.collect();
format!("[\n{}\n{}]", body.join(",\n"), indent)
}
Json::Obj(fields) => {
let body: Vec<String> = fields
.iter()
.map(|(k, v)| format!("{}{}: {}", inner, quote(k), v.pretty(&inner)))
.collect();
format!("{{\n{}\n{}}}", body.join(",\n"), indent)
}
}
}
}
fn format_number(x: f64) -> String {
if !x.is_finite() {
"null".to_string()
} else if x == x.trunc() && x.abs() < 1e15 {
format!("{}", x as i64)
} else {
format!("{}", x)
}
}
fn quote(s: &str) -> String {
let mut out = String::from("\"");
for c in s.chars() {
match c {
'"' => out.push_str("\\\""),
'\\' => out.push_str("\\\\"),
'\n' => out.push_str("\\n"),
c => out.push(c),
}
}
out.push('"');
out
}
fn mean_square(a: &[f64]) -> f64 {
a.iter().fold(0.0_f64, |s, x| s + x * x) / a.len() as f64
}
fn deviations(values: &[f64]) -> Vec<f64> {
let m = values.iter().fold(0.0_f64, |s, v| s + v) / values.len() as f64;
values.iter().map(|v| v - m).collect()
}
fn dot(a: &[f64], b: &[f64]) -> f64 {
a.iter().zip(b.iter()).fold(0.0_f64, |s, (x, y)| s + x * y)
}
fn lesson_090() -> Json {
let low_errors = [-1.0_f64, 1.0];
let high_errors = [-4.0_f64, 4.0];
let x = [1.0_f64, 2.0, 3.0, 4.0];
let duplicate: Vec<f64> = x.iter().map(|v| 2.0 * v).collect();
let (dx, dd) = (deviations(&x), deviations(&duplicate));
let r2_between = dot(&dx, &dd).powf(2.0) / (dot(&dx, &dx) * dot(&dd, &dd)); // 1 means VIF is infinite
Json::obj(vec![
("lowSpread", Json::Num(mean_square(&low_errors))),
("highSpread", Json::Num(mean_square(&high_errors))),
("exactRedundancy", Json::Bool(r2_between == 1.0)),
])
}
fn main() {
println!("{}", lesson_090().pretty(""));
}
No browser runner for Rust yet
Read the code here, then run it in your own toolchain or a ready-made cloud workspace.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}// Fintech Math Bootcamp · Lesson 090 of 120
// Regression Assumptions, Heteroskedasticity, and Multicollinearity
// Module 09: Dependence, Regression, and Model Foundations
//
// Scenario: Diagnosing unstable uncertainty and redundant predictors
// Rule: nonconstant error variance; collinearity creates unstable coefficient identification
//
// Try it: Does a robust standard error automatically solve omitted-variable confounding?
//
// Lesson article: https://thefintechbuilder.com/financial-mathematics-statistics-and-data-foundations/dependence-regression-and-model-foundations/regression-assumptions-heteroskedasticity-and-multicollinearity/
// Free course: https://courses.thefintechbuilder.com
// Synthetic teaching example, not financial advice or a production library.
using System;
using System.Collections.Generic;
using System.Linq;
using System.Text.Json;
var options = new JsonSerializerOptions { WriteIndented = true };
Console.WriteLine(JsonSerializer.Serialize(Lesson090(), options));
static double MeanSquare(double[] a) => a.Aggregate(0.0, (s, x) => s + x * x) / a.Length;
static double Average(double[] a) => a.Aggregate(0.0, (s, v) => s + v) / a.Length;
static double Dot(double[] a, double[] b) => a.Select((v, i) => v * b[i]).Aggregate(0.0, (s, p) => s + p);
static object Lesson090()
{
double[] lowErrors = { -1, 1 }, highErrors = { -4, 4 };
double[] x = { 1, 2, 3, 4 };
double[] duplicate = x.Select(v => 2 * v).ToArray();
double[] dx = x.Select(v => v - Average(x)).ToArray();
double[] dd = duplicate.Select(v => v - Average(duplicate)).ToArray();
double r2Between = Math.Pow(Dot(dx, dd), 2) / (Dot(dx, dx) * Dot(dd, dd)); // 1 means VIF is infinite
return new
{
lowSpread = MeanSquare(lowErrors),
highSpread = MeanSquare(highErrors),
exactRedundancy = r2Between == 1,
};
}
No browser runner for C# yet
Read the code here, then run it in your own toolchain or a ready-made cloud workspace.
Expected output
{
"lowSpread": 1,
"highSpread": 16,
"exactRedundancy": true
}Prefer your own machine? Every file is in the course repository · open it in Codespaces.
Lesson notes
The rule
nonconstant error variance; collinearity creates unstable coefficient identification