"""
featurevec.py -- numerical companion to the glossary entry 'feature vector'.

Every block builds a feature vector as a one-dimensional numpy array, for a
different kind of data point, and checks that the array is what the entry
says a feature vector is: a point of a Euclidean space whose dimension is
the number of features. Self-contained (numpy/matplotlib only), fixed seed.

Blocks
------
[B-house]  The entry's house offered for sale: living area, number of
           rooms and year of construction give the feature vector
           (85, 3, 1974), a point of R^3.
[B-encode] Non-numeric features of the same house: a binary feature (does
           it have a balcony) as 0 or 1, and a categorical feature with
           K = 3 values (the heating type) one-hot encoded into K entries
           of which exactly one equals 1. The house now sits in R^7.
[B-text]   A text data point (a sentence of the Universal Declaration of
           Human Rights): counting how often each word of a fixed word
           list occurs gives one entry per word.
[B-image]  An image data point: the red, green and blue intensities of its
           pixels, one entry per intensity. The feature transformation
           loses nothing here, so the image is recovered from its feature
           vector by reshaping it.
[B-audio]  An audio data point: the values its signal takes at successive
           instants, one entry per instant.
[B-dims]   The five feature vectors are points of Euclidean spaces of very
           different dimension, although one feature transformation
           produced each of them.

Outputs
-------
featurevec_audio.csv : index j and entry x_j of the audio feature vector.
featurevec_dims.csv  : kind of data point and dimension of its feature space.
featurevec.png       : preview of all five feature vectors (checking only).

Data generated by pythondemos/featurevec.py.
"""
# /// script
# requires-python = ">=3.10"
# dependencies = ["numpy==2.3.4", "matplotlib==3.10.7"]
# ///


import numpy as np
import matplotlib

matplotlib.use("Agg")
import matplotlib.pyplot as plt
from pathlib import Path

OUT_DIR = Path(__file__).parent
rng = np.random.default_rng(20261007)

report = []


def check(name, ok):
    report.append((name, bool(ok)))
    print(f"  [{'ok' if ok else 'FAIL'}] {name}")


def is_feature_vector(x, d):
    """A feature vector is a one-dimensional array of d real numbers."""
    return (isinstance(x, np.ndarray) and x.ndim == 1 and x.shape == (d,)
            and np.issubdtype(x.dtype, np.floating) and np.all(np.isfinite(x)))


# ----------------------------------------------------------------- [B-house]
print("[B-house] a house offered for sale")

house = np.array([85.0, 3.0, 1974.0])
print(f"    living area {house[0]:.0f} m^2, {house[1]:.0f} rooms, "
      f"built {house[2]:.0f}  ->  x = {house}")
check("[B-house] the house is a point of R^3", is_feature_vector(house, 3))
check("[B-house] its entries are the three features in order",
      np.array_equal(house, np.array([85.0, 3.0, 1974.0])))


# ---------------------------------------------------------------- [B-encode]
print("\n[B-encode] a binary and a categorical feature of the same house")

HEATING = ["gas", "oil", "electric"]


def encode_house(area, rooms, year, balcony, heating):
    onehot = np.zeros(len(HEATING))
    onehot[HEATING.index(heating)] = 1.0
    return np.concatenate([[area, rooms, year, float(balcony)], onehot])


house_enc = encode_house(85, 3, 1974, balcony=True, heating="oil")
onehot_part = house_enc[4:]
print(f"    balcony -> {house_enc[3]:.0f}; heating 'oil' -> {onehot_part}")
check("[B-encode] the encoded house is a point of R^7",
      is_feature_vector(house_enc, 7))
check("[B-encode] the binary feature is 0 or 1", house_enc[3] in (0.0, 1.0))
check("[B-encode] exactly one of the K = 3 one-hot entries equals 1",
      onehot_part.sum() == 1.0 and set(np.unique(onehot_part)) <= {0.0, 1.0})
check("[B-encode] the position of that entry names the value",
      HEATING[int(np.argmax(onehot_part))] == "oil")


# ------------------------------------------------------------------ [B-text]
print("\n[B-text] a text data point")

TEXT = ("All human beings are born free and equal in dignity and rights. "
        "They are endowed with reason and conscience and should act "
        "towards one another in a spirit of brotherhood.")
WORD_LIST = ["all", "and", "are", "born", "dignity", "equal",
             "free", "human", "in", "of", "reason", "rights"]


def text_features(sentence, words):
    seen = sentence.lower().replace(".", " ").replace(",", " ").split()
    return np.array([float(seen.count(w)) for w in words])


text_vec = text_features(TEXT, WORD_LIST)
print(f"    {len(WORD_LIST)} words counted; "
      f"'and' occurs {text_vec[WORD_LIST.index('and')]:.0f} times")
check("[B-text] the text is a point of R^12",
      is_feature_vector(text_vec, len(WORD_LIST)))
check("[B-text] every entry is a non-negative whole count",
      np.all(text_vec >= 0) and np.all(text_vec == np.round(text_vec)))
check("[B-text] the counts match a direct count of the words",
      text_vec.sum() == sum(TEXT.lower().replace(".", " ").replace(",", " ")
                            .split().count(w) for w in WORD_LIST))


# ----------------------------------------------------------------- [B-image]
print("\n[B-image] an image data point")

HEIGHT, WIDTH = 8, 8
rows, cols = np.mgrid[0:HEIGHT, 0:WIDTH]
# a bright disc on a darker ground, so the pixels show a shape rather than noise
disc = ((rows - 3.5) ** 2 + (cols - 3.5) ** 2) <= 2.5 ** 2
image = np.empty((HEIGHT, WIDTH, 3))
image[..., 0] = np.where(disc, 0.90, 0.15) + 0.04 * rng.normal(size=(HEIGHT, WIDTH))
image[..., 1] = np.where(disc, 0.75, 0.20) + 0.04 * rng.normal(size=(HEIGHT, WIDTH))
image[..., 2] = np.where(disc, 0.25, 0.45) + 0.04 * rng.normal(size=(HEIGHT, WIDTH))
image = np.clip(image, 0.0, 1.0)
image_vec = image.reshape(-1)
print(f"    {HEIGHT} x {WIDTH} pixels, 3 intensities each  ->  "
      f"{image_vec.size} entries")
check("[B-image] the image is a point of R^192",
      is_feature_vector(image_vec, HEIGHT * WIDTH * 3))
check("[B-image] every intensity lies between 0 and 1",
      np.all(image_vec >= 0.0) and np.all(image_vec <= 1.0))
check("[B-image] reshaping the feature vector recovers the image",
      np.array_equal(image_vec.reshape(HEIGHT, WIDTH, 3), image))


# ----------------------------------------------------------------- [B-audio]
print("\n[B-audio] an audio data point")

NR_INSTANTS = 256
instants = np.arange(NR_INSTANTS)
audio_vec = (0.6 * np.sin(2 * np.pi * 5 * instants / NR_INSTANTS)
             + 0.25 * np.sin(2 * np.pi * 17 * instants / NR_INSTANTS)
             + rng.normal(0.0, 0.03, NR_INSTANTS))
print(f"    the signal is read at {NR_INSTANTS} instants  ->  "
      f"{audio_vec.size} entries, largest magnitude "
      f"{np.abs(audio_vec).max():.2f}")
check("[B-audio] the recording is a point of R^256",
      is_feature_vector(audio_vec, NR_INSTANTS))
check("[B-audio] every entry stays within the range of the signal",
      np.all(np.abs(audio_vec) <= 1.0))
recomputed = (0.6 * np.sin(2 * np.pi * 5 * 40 / NR_INSTANTS)
              + 0.25 * np.sin(2 * np.pi * 17 * 40 / NR_INSTANTS))
check("[B-audio] entry j holds the value read at instant j",
      abs(audio_vec[40] - recomputed) < 0.15)


# ------------------------------------------------------------------ [B-dims]
print("\n[B-dims] one construction, feature spaces of very different size")

KINDS = [("house", house), ("house, encoded", house_enc),
         ("text", text_vec), ("image", image_vec), ("audio", audio_vec)]
for name, vec in KINDS:
    print(f"    {name:16s} d = {vec.size}")
dims = [vec.size for _, vec in KINDS]
check("[B-dims] each data point became a one-dimensional array",
      all(v.ndim == 1 for _, v in KINDS))
check("[B-dims] the dimensions differ by about two orders of magnitude",
      min(dims) == 3 and max(dims) == 256)

with open(OUT_DIR / "featurevec_audio.csv", "w") as fh:
    fh.write("j,x\n")
    for j, value in enumerate(audio_vec):
        fh.write(f"{j},{value:.4f}\n")

with open(OUT_DIR / "featurevec_dims.csv", "w") as fh:
    fh.write("kind,d\n")
    for name, vec in KINDS:
        fh.write(f"{name.replace(',', '')},{vec.size}\n")


# ---------------------------------------------------------------- preview
fig, axs = plt.subplots(2, 2, figsize=(9.6, 6.2))

ax_house = axs[0, 0]
labels = ["area", "rooms", "year", "balcony", "gas", "oil", "electric"]
ax_house.bar(np.arange(house_enc.size), house_enc, color="0.35")
ax_house.set_xticks(np.arange(house_enc.size))
ax_house.set_xticklabels(labels, rotation=45, ha="right", fontsize=7)
ax_house.set_yscale("symlog", linthresh=1.0)
ax_house.set_xlabel("feature")
ax_house.set_ylabel("entry value (symmetric log scale)")
ax_house.set_title("house: 3 numbers, a binary and a one-hot feature",
                   fontsize=9)

ax_text = axs[0, 1]
ax_text.bar(np.arange(text_vec.size), text_vec, color="0.35")
ax_text.set_xticks(np.arange(text_vec.size))
ax_text.set_xticklabels(WORD_LIST, rotation=60, ha="right", fontsize=7)
ax_text.set_xlabel("word of the fixed word list")
ax_text.set_ylabel("number of occurrences")
ax_text.set_title("text: one entry per word", fontsize=9)

ax_img = axs[1, 0]
ax_img.imshow(image, interpolation="nearest")
ax_img.set_xlabel("pixel column")
ax_img.set_ylabel("pixel row")
ax_img.set_title(f"image: {HEIGHT}x{WIDTH} pixels give "
                 f"{image_vec.size} entries", fontsize=9)

ax_audio = axs[1, 1]
ax_audio.plot(instants, audio_vec, "-", color="black", lw=0.9)
ax_audio.set_xlabel("instant j")
ax_audio.set_ylabel("entry $x_j$")
ax_audio.set_title(f"audio: the signal read at {NR_INSTANTS} instants",
                   fontsize=9)

fig.tight_layout()
fig.savefig(OUT_DIR / "featurevec.png", dpi=110)

n_ok = sum(ok for _, ok in report)
print(f"\n{n_ok}/{len(report)} checks pass")
print("wrote featurevec_audio.csv, featurevec_dims.csv, featurevec.png")
if n_ok != len(report):
    raise SystemExit(1)
