// =============================================================================
//  PULCINO — policy.h  (SEGNAPOSTO)
//  Questo file viene SOSTITUITO dall'esportazione dell'addestramento
//  (training/ -> policy.h, SPEC.md §5.2). Finché POLICY_VALID vale 0 il
//  firmware ignora la rete e usa la camminata CPG.
//
//  Formato: MLP 16 -> 32 -> 32 -> 6, attivazione tanh, pesi row-major W[out][in].
//  Normalizzazione: (obs - OBS_MEAN) / OBS_STD, clip ±5.
// =============================================================================
#pragma once

#define POLICY_VALID 0
#define POLICY_OBS   16
#define POLICY_H     32
#define POLICY_ACT   6

static const float OBS_MEAN[POLICY_OBS] = { 0 };
// Nota: nel segnaposto la deviazione standard è 1 (non 0) per evitare divisioni
// per zero se qualcuno forzasse POLICY_VALID a 1 senza pesi veri.
static const float OBS_STD[POLICY_OBS] = { 1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1 };

static const float W1[POLICY_H * POLICY_OBS] = { 0 };
static const float B1[POLICY_H] = { 0 };
static const float W2[POLICY_H * POLICY_H] = { 0 };
static const float B2[POLICY_H] = { 0 };
static const float W3[POLICY_ACT * POLICY_H] = { 0 };
static const float B3[POLICY_ACT] = { 0 };

// Scala dell'azione per tipo di giunto: roll, pitch, ankle (applicata a entrambe le gambe)
static const float ACTION_SCALE[3] = { 0.3f, 0.6f, 0.6f };
