Factor analysis of mixed data Table of contents Resources Data Factor analysis of mixed data is a general purpose method. It supports both numeric and categorical data.
from __future__ import annotations
import prince
dataset = prince . datasets . load_beers (). head (1000 )
dataset . head ()
is_organic style alcohol_by_volume international_bitterness_units standard_reference_method final_gravity name Lightshine Radler False Blonde 4.50 20.0 5.0 1.012 LightSwitch Lager False American Light Lager 3.95 7.5 3.0 1.005 Lightwave Belgian Pale False Belgian Pale 5.00 25.0 9.0 1.011 Like Weisse False Berlinerweisse 3.10 4.5 3.0 1.005 Lil Heaven Session IPA False Session 4.55 20.0 2.0 1.007
Fitting The rescale_with_mean and rescale_with_std parameters control whether centering and standardization are applied to the underlying PCA. By default, rescale_with_mean is True and rescale_with_std is False.
famd = prince . FAMD (
n_components = 2 ,
n_iter = 3 ,
rescale_with_mean = True ,
rescale_with_std = False ,
copy = True ,
check_input = True ,
random_state = 42 ,
engine = "sklearn" ,
handle_unknown = "error" , # same parameter as sklearn.preprocessing.OneHotEncoder
)
famd = famd . fit (dataset )
Eigenvalues eigenvalue % of variance % of variance (cumulative) component 0 3.735 3.70% 3.70% 1 1.662 1.65% 5.34%
Coordinates famd . row_coordinates (dataset ). head ()
component 0 1 name Lightshine Radler -1.795872 -0.316854 LightSwitch Lager -3.351119 -0.193896 Lightwave Belgian Pale -1.429076 -0.083288 Like Weisse -3.774585 -0.255144 Lil Heaven Session IPA -2.570021 -0.069867
component 0 1 variable alcohol_by_volume 0.920583 -0.155978 international_bitterness_units 0.815384 -0.473607 standard_reference_method 0.618738 0.621536 final_gravity 0.916577 0.158449 is_organic_False -0.000128 -0.009359 ... ... ... style_Vienna Lager -0.022171 -0.043039 style_Weizenbock 0.034561 0.092383 style_Wheat Ale -0.133955 0.002490 style_Wheatwine 0.089641 0.044065 style_Witbier -0.206367 -0.033645
103 rows × 2 columns
Visualization famd . plot (dataset , x_component = 0 , y_component = 1 )
Contributions (famd . row_contributions_ . sort_values (0 , ascending = False ). head (5 ). style . format (" {:.3%} " ))
component 0 1 name Agamemnon 0.536% 0.255% Epic Blackout Stout 0.536% 0.202% Entire Wood Aged Stout 0.536% 0.202% Agamemnon Monster Shake 0.536% 0.202% After Midnight Imperial Stout 0.536% 0.202%
famd . column_contributions_ . style . format (" {:.0%} " )
component 0 1 variable alcohol_by_volume 23% 1% international_bitterness_units 18% 13% standard_reference_method 10% 23% final_gravity 22% 2% is_organic_False 0% 0% is_organic_True 0% 0% style_Altbier 0% 1% style_Amber 0% 0% style_American Amber Lager 0% 0% style_American Barleywine 3% 1% style_American Brown 0% 1% style_American Dark Lager 0% 0% style_American IPA 1% 7% style_American Imperial Porter 0% 0% style_American Imperial Stout 5% 4% style_American Lager 1% 0% style_American Light Lager 1% 0% style_American Low-Carb Lager 0% 0% style_American Oktoberfest 0% 0% style_American Pale 1% 0% style_American Pilsener 0% 0% style_American Premium Lager 0% 0% style_American Stout 1% 2% style_American Strong Pale 0% 0% style_Austrailian Pale 0% 0% style_Baltic Porter 0% 1% style_Belgian Blonde 0% 0% style_Belgian Dark Strong 0% 0% style_Belgian Dubbel 0% 0% style_Belgian Pale 0% 0% style_Belgian Pale Strong 0% 0% style_Belgian Quad 0% 0% style_Belgian Table Beer 0% 0% style_Belgian Tripel 0% 0% style_Berlinerweisse 2% 0% style_Bernsteinfarbenesweizen 0% 0% style_Bitter 0% 1% style_Bière de Garde 0% 0% style_Black Ale 0% 0% style_Blonde 1% 0% style_Bock 0% 1% style_Bohemian Pilsener 0% 0% style_British Barleywine 0% 0% style_British Imperial Stout 0% 1% style_Brown Porter 0% 1% style_Chocolate Beer 0% 0% style_Coffee Beer 0% 0% style_Cream Ale 1% 0% style_Doppelbock 0% 1% style_Dortmunder 0% 0% style_Dry Irish Stout 0% 0% style_Dunkelweizen 0% 0% style_ESB 0% 0% style_English Brown 0% 0% style_English Dark Mild 0% 0% style_English IPA 0% 0% style_English Pale 0% 0% style_English Pale Mild 0% 0% style_English Summer Ale 0% 0% style_Euro Dark 0% 0% style_Export Stout 0% 0% style_Flanders Red 0% 1% style_Fruit Beer 0% 2% style_Fruit Wheat Ale 0% 0% style_German Pilsener 0% 0% style_Hefeweizen 0% 0% style_Helles 0% 0% style_Imperial IPA 3% 15% style_Imperial Red 0% 2% style_International Pale 0% 0% style_International Pilsener 0% 0% style_Irish Red 0% 1% style_Juicy or Hazy IPA 0% 0% style_Juicy or Hazy Pale Ale 0% 0% style_Kölsch 0% 0% style_Leichtbier 0% 0% style_Leichtesweizen 0% 0% style_Leipzig Gose 0% 1% style_Maibock 0% 0% style_Malt Liquor 0% 0% style_Märzen 0% 0% style_Oatmeal Stout 0% 1% style_Oktoberfest 0% 0% style_Old Ale 0% 0% style_Pumpkin Beer 0% 1% style_Robust Porter 0% 1% style_Saison 0% 0% style_Schwarzbier 0% 0% style_Scotch Ale 0% 1% style_Scottish Heavy 0% 0% style_Scottish Light 0% 0% style_Session 0% 0% style_Smoke Porter 0% 0% style_Special Bitter 0% 0% style_Spice Beer 0% 2% style_Stout 1% 1% style_Strong Ale 0% 0% style_Sweet Stout 0% 3% style_Vienna Lager 0% 0% style_Weizenbock 0% 1% style_Wheat Ale 0% 0% style_Wheatwine 0% 0% style_Witbier 1% 0%
Wikipedia example The Wikipedia article on FAMD uses a small dataset with 6 individuals, 3 quantitative variables (k₁, k₂, k₃) and 3 qualitative variables (q₁, q₂, q₃). Let’s reproduce it.
import pandas as pd
wiki = pd . DataFrame (
{
"k1" : [2.0 , 5.0 , 3.0 , 4.0 , 1.0 , 6.0 ],
"k2" : [4.5 , 4.5 , 1.0 , 1.0 , 1.0 , 1.0 ],
"k3" : [4.0 , 4.0 , 2.0 , 2.0 , 1.0 , 2.0 ],
"q1" : ["A" , "C" , "B" , "B" , "A" , "C" ],
"q2" : ["B" , "B" , "B" , "B" , "A" , "A" ],
"q3" : ["C" , "C" , "B" , "B" , "A" , "A" ],
},
index = [f "i { i } " for i in range (1 , 7 )],
)
wiki
k1 k2 k3 q1 q2 q3 i1 2.0 4.5 4.0 A B C i2 5.0 4.5 4.0 C B C i3 3.0 1.0 2.0 B B B i4 4.0 1.0 2.0 B B B i5 1.0 1.0 1.0 A A A i6 6.0 1.0 2.0 C A A
wiki_famd = prince . FAMD (n_components = 5 , engine = "scipy" )
wiki_famd = wiki_famd . fit (wiki )
wiki_famd . eigenvalues_summary
eigenvalue % of variance % of variance (cumulative) component 0 3.467 43.34% 43.34% 1 2.500 31.25% 74.59% 2 1.971 24.64% 99.22% 3 0.056 0.70% 99.92% 4 0.006 0.08% 100.00%
import altair as alt
rc = wiki_famd . row_coordinates (wiki )
rc . columns = [f "F { c + 1 } " for c in rc . columns ]
rc ["label" ] = rc . index
alt . Chart (rc . reset_index (drop = True )). mark_text (fontSize = 14 ). encode (
x = alt . X ("F1:Q" , title = "F1" ),
y = alt . Y ("F2:Q" , title = "F2" ),
text = "label:N" ,
). properties (title = "Figure 1 — Individuals" , width = 400 , height = 300 )
num_cols = ["k1" , "k2" , "k3" ]
cat_cols = ["q1" , "q2" , "q3" ]
# FactoMineR's var$coord: r² for numerical, η² for categorical (one row per variable).
cc = wiki_famd . variable_coordinates_ . iloc [:, :2 ]. copy ()
cc . columns = ["F1" , "F2" ]
cc ["variable" ] = cc . index
cc ["type" ] = ["quanti" ] * len (num_cols ) + ["quali" ] * len (cat_cols )
points = (
alt . Chart (cc . reset_index (drop = True ))
. mark_text (fontSize = 14 )
. encode (
x = alt . X ("F1:Q" , title = "F1 (r² / η²)" , scale = alt . Scale (domain = [0 , 1.05 ])),
y = alt . Y ("F2:Q" , title = "F2 (r² / η²)" , scale = alt . Scale (domain = [0 , 1.05 ])),
text = "variable:N" ,
color = alt . Color ("type:N" , legend = alt . Legend (title = "Variable type" )),
)
)
points . properties (title = "Figure 2 — Relationship square" , width = 400 , height = 400 )
import numpy as np
num_cols = ["k1" , "k2" , "k3" ]
cat_cols = ["q1" , "q2" , "q3" ]
corr_circle = wiki_famd . column_correlations . loc [num_cols ]. iloc [:, :2 ]. copy ()
corr_circle . columns = ["F1" , "F2" ]
corr_circle ["variable" ] = corr_circle . index
# Arrows from origin to each variable
arrows = pd . DataFrame (
[
{"x" : 0 , "y" : 0 , "x2" : row ["F1" ], "y2" : row ["F2" ], "variable" : row ["variable" ]}
for _ , row in corr_circle . iterrows ()
]
)
circle_theta = np . linspace (0 , 2 * np . pi , 100 )
unit_circle = pd . DataFrame (
{"x" : np . cos (circle_theta ), "y" : np . sin (circle_theta ), "order" : range (100 )}
)
c_circle = (
alt . Chart (unit_circle )
. mark_line (color = "gray" , strokeDash = [4 , 4 ])
. encode (
x = alt . X ("x:Q" , title = "F1" , scale = alt . Scale (domain = [- 1.2 , 1.2 ])),
y = alt . Y ("y:Q" , title = "F2" , scale = alt . Scale (domain = [- 1.2 , 1.2 ])),
order = "order:O" ,
)
)
c_arrows = (
alt . Chart (arrows )
. mark_rule (strokeWidth = 2 )
. encode (
x = "x:Q" ,
y = "y:Q" ,
x2 = "x2:Q" ,
y2 = "y2:Q" ,
color = "variable:N" ,
)
)
c_labels = (
alt . Chart (corr_circle . reset_index (drop = True ))
. mark_text (fontSize = 14 , dx = 15 )
. encode (x = "F1:Q" , y = "F2:Q" , text = "variable:N" )
)
(c_circle + c_arrows + c_labels ). properties (
title = "Figure 3 — Correlation circle" , width = 400 , height = 400
)
# Category coordinates are the mean of row coordinates for each category
cat_coords = []
for q in cat_cols :
for cat in sorted (wiki [q ]. unique ()):
mean_rc = rc . loc [wiki [q ] == cat , ["F1" , "F2" ]]. mean ()
cat_coords . append ({"F1" : mean_rc ["F1" ], "F2" : mean_rc ["F2" ], "label" : f " { q } = { cat } " })
cat_df = pd . DataFrame (cat_coords )
alt . Chart (cat_df ). mark_text (fontSize = 14 ). encode (
x = alt . X ("F1:Q" , title = "F1" ),
y = alt . Y ("F2:Q" , title = "F2" ),
text = "label:N" ,
). properties (title = "Figure 4 — Categories" , width = 400 , height = 300 )