diff --git a/MyIA.AI.Notebooks/GenAI/FineTuning/FT-00b-LoRA-Hyperparams-from-scratch.ipynb b/MyIA.AI.Notebooks/GenAI/FineTuning/FT-00b-LoRA-Hyperparams-from-scratch.ipynb index 2c629ff64e..396a4bcf9a 100644 --- a/MyIA.AI.Notebooks/GenAI/FineTuning/FT-00b-LoRA-Hyperparams-from-scratch.ipynb +++ b/MyIA.AI.Notebooks/GenAI/FineTuning/FT-00b-LoRA-Hyperparams-from-scratch.ipynb @@ -5,10 +5,10 @@ "id": "aad8b69c", "metadata": { "papermill": { - "duration": 0.003325, - "end_time": "2026-09-14T02:48:43.510422", + "duration": 0.005715, + "end_time": "2026-09-14T10:39:54.602528", "exception": false, - "start_time": "2026-09-14T02:48:43.507097", + "start_time": "2026-09-14T10:39:54.596813", "status": "completed" }, "tags": [] @@ -30,10 +30,10 @@ "id": "b42fee96", "metadata": { "papermill": { - "duration": 0.003, - "end_time": "2026-09-14T02:48:43.516423", + "duration": 0.004671, + "end_time": "2026-09-14T10:39:54.615867", "exception": false, - "start_time": "2026-09-14T02:48:43.513423", + "start_time": "2026-09-14T10:39:54.611196", "status": "completed" }, "tags": [] @@ -48,16 +48,16 @@ "id": "143c50c3", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:48:43.523697Z", - "iopub.status.busy": "2026-09-14T02:48:43.523697Z", - "iopub.status.idle": "2026-09-14T02:48:46.774044Z", - "shell.execute_reply": "2026-09-14T02:48:46.773534Z" + "iopub.execute_input": "2026-09-14T10:39:54.628901Z", + "iopub.status.busy": "2026-09-14T10:39:54.627893Z", + "iopub.status.idle": "2026-09-14T10:39:57.296027Z", + "shell.execute_reply": "2026-09-14T10:39:57.296027Z" }, "papermill": { - "duration": 3.254113, - "end_time": "2026-09-14T02:48:46.774044", + "duration": 2.676166, + "end_time": "2026-09-14T10:39:57.297033", "exception": false, - "start_time": "2026-09-14T02:48:43.519931", + "start_time": "2026-09-14T10:39:54.620867", "status": "completed" }, "tags": [] @@ -92,10 +92,10 @@ "id": "cae59fd9", "metadata": { "papermill": { - "duration": 0.003307, - "end_time": "2026-09-14T02:48:46.780523", + "duration": 0.003008, + "end_time": "2026-09-14T10:39:57.303039", "exception": false, - "start_time": "2026-09-14T02:48:46.777216", + "start_time": "2026-09-14T10:39:57.300031", "status": "completed" }, "tags": [] @@ -111,10 +111,10 @@ "id": "eaad34b5", "metadata": { "papermill": { - "duration": 0.002522, - "end_time": "2026-09-14T02:48:46.786166", + "duration": 0.003, + "end_time": "2026-09-14T10:39:57.309041", "exception": false, - "start_time": "2026-09-14T02:48:46.783644", + "start_time": "2026-09-14T10:39:57.306041", "status": "completed" }, "tags": [] @@ -131,16 +131,16 @@ "id": "906e24f3", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:48:46.793888Z", - "iopub.status.busy": "2026-09-14T02:48:46.793377Z", - "iopub.status.idle": "2026-09-14T02:48:46.799385Z", - "shell.execute_reply": "2026-09-14T02:48:46.798825Z" + "iopub.execute_input": "2026-09-14T10:39:57.317070Z", + "iopub.status.busy": "2026-09-14T10:39:57.317070Z", + "iopub.status.idle": "2026-09-14T10:39:57.321699Z", + "shell.execute_reply": "2026-09-14T10:39:57.321699Z" }, "papermill": { - "duration": 0.011158, - "end_time": "2026-09-14T02:48:46.799900", + "duration": 0.010671, + "end_time": "2026-09-14T10:39:57.322712", "exception": false, - "start_time": "2026-09-14T02:48:46.788742", + "start_time": "2026-09-14T10:39:57.312041", "status": "completed" }, "tags": [] @@ -193,10 +193,10 @@ "id": "96ced9d9", "metadata": { "papermill": { - "duration": 0.002515, - "end_time": "2026-09-14T02:48:46.805075", + "duration": 0.003009, + "end_time": "2026-09-14T10:39:57.329092", "exception": false, - "start_time": "2026-09-14T02:48:46.802560", + "start_time": "2026-09-14T10:39:57.326083", "status": "completed" }, "tags": [] @@ -212,10 +212,10 @@ "id": "601a04a2", "metadata": { "papermill": { - "duration": 0.003, - "end_time": "2026-09-14T02:48:46.811089", + "duration": 0.003029, + "end_time": "2026-09-14T10:39:57.334665", "exception": false, - "start_time": "2026-09-14T02:48:46.808089", + "start_time": "2026-09-14T10:39:57.331636", "status": "completed" }, "tags": [] @@ -232,335 +232,27 @@ "id": "e4435a7e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:48:46.818136Z", - "iopub.status.busy": "2026-09-14T02:48:46.817602Z", - "iopub.status.idle": "2026-09-14T02:48:50.897382Z", - "shell.execute_reply": "2026-09-14T02:48:50.897382Z" + "iopub.execute_input": "2026-09-14T10:39:57.342288Z", + "iopub.status.busy": "2026-09-14T10:39:57.341658Z", + "iopub.status.idle": "2026-09-14T10:39:57.377457Z", + "shell.execute_reply": "2026-09-14T10:39:57.377457Z" }, "papermill": { - "duration": 4.085294, - "end_time": "2026-09-14T02:48:50.899402", + "duration": 0.040797, + "end_time": "2026-09-14T10:39:57.378461", "exception": false, - "start_time": "2026-09-14T02:48:46.814108", + "start_time": "2026-09-14T10:39:57.337664", "status": "completed" }, "tags": [] }, "outputs": [ - { - "name": "stderr", - "output_type": "stream", - "text": [ - "\r", - " 0%| | 0.00/26.4M [00:00 522 params (full head = 5130)\n", @@ -1087,10 +779,10 @@ "id": "5ea73a76", "metadata": { "papermill": { - "duration": 0.007512, - "end_time": "2026-09-14T02:50:13.261273", + "duration": 0.003999, + "end_time": "2026-09-14T10:40:49.577599", "exception": false, - "start_time": "2026-09-14T02:50:13.253761", + "start_time": "2026-09-14T10:40:49.573600", "status": "completed" }, "tags": [] @@ -1106,10 +798,10 @@ "id": "a7549c09", "metadata": { "papermill": { - "duration": 0.01008, - "end_time": "2026-09-14T02:50:13.277276", + "duration": 0.004, + "end_time": "2026-09-14T10:40:49.586599", "exception": false, - "start_time": "2026-09-14T02:50:13.267196", + "start_time": "2026-09-14T10:40:49.582599", "status": "completed" }, "tags": [] @@ -1130,16 +822,16 @@ "id": "c2d8dbd5", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:50:13.296604Z", - "iopub.status.busy": "2026-09-14T02:50:13.296083Z", - "iopub.status.idle": "2026-09-14T02:50:16.807782Z", - "shell.execute_reply": "2026-09-14T02:50:16.806758Z" + "iopub.execute_input": "2026-09-14T10:40:49.596845Z", + "iopub.status.busy": "2026-09-14T10:40:49.596845Z", + "iopub.status.idle": "2026-09-14T10:40:52.297714Z", + "shell.execute_reply": "2026-09-14T10:40:52.296700Z" }, "papermill": { - "duration": 3.523419, - "end_time": "2026-09-14T02:50:16.808845", + "duration": 2.705973, + "end_time": "2026-09-14T10:40:52.297714", "exception": false, - "start_time": "2026-09-14T02:50:13.285426", + "start_time": "2026-09-14T10:40:49.591741", "status": "completed" }, "tags": [] @@ -1189,10 +881,10 @@ "id": "454b3785", "metadata": { "papermill": { - "duration": 0.006954, - "end_time": "2026-09-14T02:50:16.823179", + "duration": 0.007512, + "end_time": "2026-09-14T10:40:52.312823", "exception": false, - "start_time": "2026-09-14T02:50:16.816225", + "start_time": "2026-09-14T10:40:52.305311", "status": "completed" }, "tags": [] @@ -1212,10 +904,10 @@ "id": "2a24e919", "metadata": { "papermill": { - "duration": 0.0089, - "end_time": "2026-09-14T02:50:16.842686", + "duration": 0.006015, + "end_time": "2026-09-14T10:40:52.326437", "exception": false, - "start_time": "2026-09-14T02:50:16.833786", + "start_time": "2026-09-14T10:40:52.320422", "status": "completed" }, "tags": [] @@ -1231,10 +923,10 @@ "id": "46350b37", "metadata": { "papermill": { - "duration": 0.008599, - "end_time": "2026-09-14T02:50:16.859789", + "duration": 0.006743, + "end_time": "2026-09-14T10:40:52.340185", "exception": false, - "start_time": "2026-09-14T02:50:16.851190", + "start_time": "2026-09-14T10:40:52.333442", "status": "completed" }, "tags": [] @@ -1251,16 +943,264 @@ "id": "a765dfff", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:50:16.875676Z", - "iopub.status.busy": "2026-09-14T02:50:16.875676Z", - "iopub.status.idle": "2026-09-14T02:50:16.881365Z", - "shell.execute_reply": "2026-09-14T02:50:16.880360Z" + "iopub.execute_input": "2026-09-14T10:40:52.355168Z", + "iopub.status.busy": "2026-09-14T10:40:52.355168Z", + "iopub.status.idle": "2026-09-14T10:40:52.359168Z", + "shell.execute_reply": "2026-09-14T10:40:52.359168Z" + }, + "papermill": { + "duration": 0.012005, + "end_time": "2026-09-14T10:40:52.359168", + "exception": false, + "start_time": "2026-09-14T10:40:52.347163", + "status": "completed" + }, + "tags": [] + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Exercice a completer : construire ratios et identifier best.\n" + ] + } + ], + "source": [ + "# Exercice 1 : ratio params / accuracy sur alpha/r = 1.0\n", + "#\n", + "# Indice 1 : `results` est une liste de tuples (r, scaling, alpha, n_params, acc_test)\n", + "# deja construite par les cellules precedentes. Filtrer sur scaling == 1.0.\n", + "# Indice 2 : pour chaque r retenu, calculer `n_params / acc_test` (proteger le diviseur\n", + "# par `max(acc, 1e-6)` si une exactitude est nulle).\n", + "# Indice 3 : le « juste rang » est le r qui MINIMISE ce ratio. Commentez en une phrase\n", + "# pourquoi le rapport n'est PAS monotone en r.\n", + "#\n", + "# Etape 1 : construire la liste `ratios = [(r, n, acc, ratio), ...]`\n", + "# Etape 2 : identifier `best = min(ratios, key=...)` et l'imprimer\n", + "\n", + "ratios = [] # TODO etudiant\n", + "best = None # TODO etudiant\n", + "print(\"Exercice a completer : construire ratios et identifier best.\")\n" + ] + }, + { + "cell_type": "markdown", + "id": "9c657101", + "metadata": { + "papermill": { + "duration": 0.005002, + "end_time": "2026-09-14T10:40:52.371068", + "exception": false, + "start_time": "2026-09-14T10:40:52.366066", + "status": "completed" + }, + "tags": [] + }, + "source": [ + "### Exercice 2 : pourquoi `alpha/r = 2.0` dégrade-t-il parfois ?\n", + "\n", + "Reprendre l'invariant de FT-00a : `‖delta_W‖ = ‖(alpha/r) * B @ A‖`. Pour `r = 4` et `alpha/r = 2.0`, mesurer `‖B @ A‖` à la **fin** de l'entraînement et comparer à `alpha/r = 1.0`. Le produit `alpha/r * ‖B @ A‖` borne la magnitude du delta — expliquez en deux lignes pourquoi un delta trop grand peut **déstabiliser** la cible." + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "id": "d5b6ff9f", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-14T10:40:52.380581Z", + "iopub.status.busy": "2026-09-14T10:40:52.380581Z", + "iopub.status.idle": "2026-09-14T10:40:52.385389Z", + "shell.execute_reply": "2026-09-14T10:40:52.385389Z" + }, + "papermill": { + "duration": 0.011233, + "end_time": "2026-09-14T10:40:52.386395", + "exception": false, + "start_time": "2026-09-14T10:40:52.375162", + "status": "completed" + }, + "tags": [] + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Exercice a completer : definir train_and_measure et boucler les deux scalings.\n" + ] + } + ], + "source": [ + "# Exercice 2 : ||B @ A|| a la fin de l'entrainement pour les deux scalings\n", + "#\n", + "# Indice 1 : reprendre la structure de la cellule « mesure des reguliers » au-dessus.\n", + "# Pour chaque scaling, recreer un modele + tete LoRA(r=4, alpha=scaling*r),\n", + "# copier base_model, geler le backbone, instancier un optimiseur Adam sur\n", + "# les poids de tete uniquement.\n", + "# Indice 2 : apres `epochs` passes sur train_loader, lire `m.head.scaling * (m.head.B @ m.head.A)`,\n", + "# calculer sa norme Frobenius : `delta_W.norm().item()`.\n", + "# Indice 3 : comparer `alpha/r * ||B @ A||` entre les deux scalings. Un delta plus grand\n", + "# pour `alpha/r = 2.0` explique pourquoi ce scaling destabilise parfois.\n", + "#\n", + "# Etape 1 : definir `def train_and_measure(r, scaling, epochs=2):`\n", + "# retourner (delta_W.norm().item(), A.norm().item(), B.norm().item())\n", + "# Etape 2 : boucler `for scaling in [1.0, 2.0]:` et imprimer les trois normes\n", + "\n", + "def train_and_measure(r, scaling, epochs=2):\n", + " \"\"\"Mesure ||B@A|| et normes A, B apres entrainement.\"\"\"\n", + " pass # TODO etudiant\n", + "\n", + "\n", + "# for scaling in [1.0, 2.0]:\n", + "# norm_dW, norm_A, norm_B = train_and_measure(r=4, scaling=scaling)\n", + "# print(f\"alpha/r={scaling} ||B@A||={norm_dW:.4f} ||A||={norm_A:.4f} ||B||={norm_B:.4f}\")\n", + "print(\"Exercice a completer : definir train_and_measure et boucler les deux scalings.\")\n" + ] + }, + { + "cell_type": "markdown", + "id": "07db3164", + "metadata": { + "papermill": { + "duration": 0.006013, + "end_time": "2026-09-14T10:40:52.396518", + "exception": false, + "start_time": "2026-09-14T10:40:52.390505", + "status": "completed" + }, + "tags": [] + }, + "source": [ + "### Exercice 3 : seuil d'amplification — combien d'epochs pour diverger ?\n", + "\n", + "Pour `r = 4` et `alpha/r = 2.0`, entraîner **6 epochs** (au lieu de 2). À chaque epoch, mesurer l'exactitude test. Identifier l'epoch où l'exactitude **plafonne ou redescend** — c'est le seuil au-delà duquel le scaling amplifié devient contre-productif." + ] + }, + { + "cell_type": "code", + "execution_count": 11, + "id": "910d1c53", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-14T10:40:52.411636Z", + "iopub.status.busy": "2026-09-14T10:40:52.410020Z", + "iopub.status.idle": "2026-09-14T10:40:52.414970Z", + "shell.execute_reply": "2026-09-14T10:40:52.414970Z" }, "papermill": { - "duration": 0.01616, - "end_time": "2026-09-14T02:50:16.881365", + "duration": 0.013503, + "end_time": "2026-09-14T10:40:52.416016", "exception": false, - "start_time": "2026-09-14T02:50:16.865205", + "start_time": "2026-09-14T10:40:52.402513", + "status": "completed" + }, + "tags": [] + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "epoch | acc test\n", + "------|----------\n", + "Exercice a completer : boucle epoch et accumulation des exactitudes test.\n" + ] + } + ], + "source": [ + "# Exercice 3 : seuil d'amplification sur 6 epochs (r=4, alpha/r = 2.0)\n", + "#\n", + "# Indice 1 : recopier la cellule de comparaison des scalings au-dessus, en fixant\n", + "# r=4 et alpha=8.0. Ajouter une boucle d'epochs : for epoch in range(1, 7).\n", + "# Indice 2 : a chaque epoch, appeler `evaluate(m, test_loader, transform=None)` et\n", + "# imprimer `f\"{epoch} | {acc:.4f}\"` au format table.\n", + "# Indice 3 : reperer l'epoch ou l'exactitude plafonne OU redescend — c'est le seuil\n", + "# au-dela duquel `alpha/r = 2.0` devient contre-productif.\n", + "#\n", + "# Etape 1 : instancier le modele + tete LoRA(r=4, alpha=8.0)\n", + "# Etape 2 : Adam sur la tete\n", + "# Etape 3 : boucle epoch, eval a chaque pas, imprimer la table\n", + "\n", + "epoch_acc = [] # TODO etudiant : liste de tuples (epoch, acc_test)\n", + "print(\"epoch | acc test\")\n", + "print(\"------|----------\")\n", + "# for epoch in range(1, 7):\n", + "# ...\n", + "# epoch_acc.append((epoch, acc))\n", + "# print(f\" {epoch} | {acc:.4f}\")\n", + "print(\"Exercice a completer : boucle epoch et accumulation des exactitudes test.\")\n" + ] + }, + { + "cell_type": "markdown", + "id": "11a7d134", + "metadata": { + "papermill": { + "duration": 0.004073, + "end_time": "2026-09-14T10:40:52.426912", + "exception": false, + "start_time": "2026-09-14T10:40:52.422839", + "status": "completed" + }, + "tags": [] + }, + "source": [ + "## Résumé\n", + "\n", + "Trois régularités mesurées sur la même mini-tâche, six valeurs de rang, trois scalings :\n", + "\n", + "- **`alpha/r = 1.0`** est le choix de référence : la magnitude effective du signal est neutre, et le rang commande seul la dimensionnalité.\n", + "- **`r = 1` est trop petit** sur Fashion-MNIST inversé : un seul degré de liberté sous-adapte. Au-delà de `r = 4`, les gains saturent.\n", + "- **`alpha/r = 2.0`** peut dégrader : trop de signal amplifié, oscillations ou divergence sur les grands modèles.\n", + "\n", + "Le découplage `alpha / r` est ce qui rend les deux knobs **réglables indépendamment** : augmenter `r` ne change pas la magnitude du signal, et changer `alpha` sans toucher `r` ne change pas la dimensionnalité. C'est l'ingrédient qui rend LoRA **praticable** sur des modèles 7B+ — où le rang reste petit (`r = 8` typique) mais le scaling devient un hyperparamètre de calibration fin.\n", + "\n", + "**Branchement** : FT-00a (mécanisme) → FT-00b (réglage) → FT-01 (LoRA avec `peft` sur GPT-2) → FT-02 (QLoRA 4-bit) → FT-06 (vision-langage)." + ] + }, + { + "cell_type": "markdown", + "id": "81a4784d", + "metadata": { + "papermill": { + "duration": 0.004107, + "end_time": "2026-09-14T10:40:52.435919", + "exception": false, + "start_time": "2026-09-14T10:40:52.431812", + "status": "completed" + }, + "tags": [] + }, + "source": [ + "## Corrigés\n", + "\n", + "Les solutions ci-dessous reprennent le code exact des trois cellules d'origine,\n", + "déplacé ici pour préserver la valeur pédagogique de la mesure (chiffres réels\n", + "sur Fashion-MNIST inversé). Elles sont **référencées** par les énoncés ci-dessus —\n", + "un étudiant les consulte **après** avoir tenté l'exercice, jamais avant.\n", + "\n", + "Pour rejouer les solutions : exécuter cette section. Les sorties sont bit-identiques\n", + "à celles des cellules d'origine (mêmes seeds, mêmes hyperparamètres).\n" + ] + }, + { + "cell_type": "code", + "execution_count": 12, + "id": "fc0c9cc0", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-14T10:40:52.445813Z", + "iopub.status.busy": "2026-09-14T10:40:52.445813Z", + "iopub.status.idle": "2026-09-14T10:40:52.450851Z", + "shell.execute_reply": "2026-09-14T10:40:52.450851Z" + }, + "papermill": { + "duration": 0.011044, + "end_time": "2026-09-14T10:40:52.451856", + "exception": false, + "start_time": "2026-09-14T10:40:52.440812", "status": "completed" }, "tags": [] @@ -1271,18 +1211,20 @@ "output_type": "stream", "text": [ "=== ratio params / accuracy (alpha/r = 1.0) ===\n", - " r= 1 params= 522 acc=0.2100 params/acc= 2485.7\n", + " r= 1 params= 522 acc=0.2095 params/acc= 2491.6\n", " r= 2 params= 1044 acc=0.2440 params/acc= 4278.7\n", " r= 4 params= 2088 acc=0.3210 params/acc= 6504.7\n", " r= 8 params= 4176 acc=0.5345 params/acc= 7812.9\n", " r=16 params= 8352 acc=0.5765 params/acc= 14487.4\n", " r=32 params=16704 acc=0.6190 params/acc= 26985.5\n", "\n", - "Juste rang (min params/acc) : r=1, acc=0.2100, ratio=2485.7\n" + "Juste rang (min params/acc) : r=1, acc=0.2095, ratio=2491.6\n" ] } ], "source": [ + "# Corrige Exercice 1\n", + "\n", "# Exercice 1\n", "print(\"=== ratio params / accuracy (alpha/r = 1.0) ===\")\n", "ratios = []\n", @@ -1301,41 +1243,22 @@ " print(\"Aucune donnee pour ce scaling.\")" ] }, - { - "cell_type": "markdown", - "id": "9c657101", - "metadata": { - "papermill": { - "duration": 0.007207, - "end_time": "2026-09-14T02:50:16.894086", - "exception": false, - "start_time": "2026-09-14T02:50:16.886879", - "status": "completed" - }, - "tags": [] - }, - "source": [ - "### Exercice 2 : pourquoi `alpha/r = 2.0` dégrade-t-il parfois ?\n", - "\n", - "Reprendre l'invariant de FT-00a : `‖delta_W‖ = ‖(alpha/r) * B @ A‖`. Pour `r = 4` et `alpha/r = 2.0`, mesurer `‖B @ A‖` à la **fin** de l'entraînement et comparer à `alpha/r = 1.0`. Le produit `alpha/r * ‖B @ A‖` borne la magnitude du delta — expliquez en deux lignes pourquoi un delta trop grand peut **déstabiliser** la cible." - ] - }, { "cell_type": "code", - "execution_count": 10, - "id": "d5b6ff9f", + "execution_count": 13, + "id": "282d9035", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:50:16.905860Z", - "iopub.status.busy": "2026-09-14T02:50:16.904119Z", - "iopub.status.idle": "2026-09-14T02:50:24.101768Z", - "shell.execute_reply": "2026-09-14T02:50:24.100736Z" + "iopub.execute_input": "2026-09-14T10:40:52.461552Z", + "iopub.status.busy": "2026-09-14T10:40:52.461552Z", + "iopub.status.idle": "2026-09-14T10:40:57.497123Z", + "shell.execute_reply": "2026-09-14T10:40:57.496611Z" }, "papermill": { - "duration": 7.203644, - "end_time": "2026-09-14T02:50:24.102763", + "duration": 5.041273, + "end_time": "2026-09-14T10:40:57.498128", "exception": false, - "start_time": "2026-09-14T02:50:16.899119", + "start_time": "2026-09-14T10:40:52.456855", "status": "completed" }, "tags": [] @@ -1345,18 +1268,20 @@ "name": "stdout", "output_type": "stream", "text": [ - "alpha/r=1.0 ||B@A||=1.2087 ||A||=3.9232 ||B||=0.4396\n" + "alpha/r=1.0 ||B@A||=1.2088 ||A||=3.9234 ||B||=0.4396\n" ] }, { "name": "stdout", "output_type": "stream", "text": [ - "alpha/r=2.0 ||B@A||=1.7818 ||A||=3.6063 ||B||=0.3849\n" + "alpha/r=2.0 ||B@A||=1.7821 ||A||=3.6065 ||B||=0.3850\n" ] } ], "source": [ + "# Corrige Exercice 2\n", + "\n", "# Exercice 2 -- on re-entraine brievement r=4 pour les deux scalings et on lit ||B @ A||\n", "def train_and_measure(r, scaling, epochs=2):\n", " alpha = scaling * r\n", @@ -1390,41 +1315,22 @@ " )" ] }, - { - "cell_type": "markdown", - "id": "07db3164", - "metadata": { - "papermill": { - "duration": 0.008761, - "end_time": "2026-09-14T02:50:24.121948", - "exception": false, - "start_time": "2026-09-14T02:50:24.113187", - "status": "completed" - }, - "tags": [] - }, - "source": [ - "### Exercice 3 : seuil d'amplification — combien d'epochs pour diverger ?\n", - "\n", - "Pour `r = 4` et `alpha/r = 2.0`, entraîner **6 epochs** (au lieu de 2). À chaque epoch, mesurer l'exactitude test. Identifier l'epoch où l'exactitude **plafonne ou redescend** — c'est le seuil au-delà duquel le scaling amplifié devient contre-productif." - ] - }, { "cell_type": "code", - "execution_count": 11, - "id": "910d1c53", + "execution_count": 14, + "id": "939fdddf", "metadata": { "execution": { - "iopub.execute_input": "2026-09-14T02:50:24.140660Z", - "iopub.status.busy": "2026-09-14T02:50:24.139645Z", - "iopub.status.idle": "2026-09-14T02:50:37.614164Z", - "shell.execute_reply": "2026-09-14T02:50:37.613156Z" + "iopub.execute_input": "2026-09-14T10:40:57.509753Z", + "iopub.status.busy": "2026-09-14T10:40:57.509753Z", + "iopub.status.idle": "2026-09-14T10:41:05.742915Z", + "shell.execute_reply": "2026-09-14T10:41:05.742387Z" }, "papermill": { - "duration": 13.484188, - "end_time": "2026-09-14T02:50:37.615164", + "duration": 8.240761, + "end_time": "2026-09-14T10:41:05.744496", "exception": false, - "start_time": "2026-09-14T02:50:24.130976", + "start_time": "2026-09-14T10:40:57.503735", "status": "completed" }, "tags": [] @@ -1477,11 +1383,13 @@ "name": "stdout", "output_type": "stream", "text": [ - " 6 | 0.6155\n" + " 6 | 0.6150\n" ] } ], "source": [ + "# Corrige Exercice 3\n", + "\n", "# Exercice 3\n", "torch.manual_seed(SEED)\n", "m = copy.deepcopy(base_model)\n", @@ -1507,33 +1415,6 @@ " acc = evaluate(m, test_loader, transform=None)\n", " print(f\" {epoch} | {acc:.4f}\")" ] - }, - { - "cell_type": "markdown", - "id": "11a7d134", - "metadata": { - "papermill": { - "duration": 0.008669, - "end_time": "2026-09-14T02:50:37.634494", - "exception": false, - "start_time": "2026-09-14T02:50:37.625825", - "status": "completed" - }, - "tags": [] - }, - "source": [ - "## Résumé\n", - "\n", - "Trois régularités mesurées sur la même mini-tâche, six valeurs de rang, trois scalings :\n", - "\n", - "- **`alpha/r = 1.0`** est le choix de référence : la magnitude effective du signal est neutre, et le rang commande seul la dimensionnalité.\n", - "- **`r = 1` est trop petit** sur Fashion-MNIST inversé : un seul degré de liberté sous-adapte. Au-delà de `r = 4`, les gains saturent.\n", - "- **`alpha/r = 2.0`** peut dégrader : trop de signal amplifié, oscillations ou divergence sur les grands modèles.\n", - "\n", - "Le découplage `alpha / r` est ce qui rend les deux knobs **réglables indépendamment** : augmenter `r` ne change pas la magnitude du signal, et changer `alpha` sans toucher `r` ne change pas la dimensionnalité. C'est l'ingrédient qui rend LoRA **praticable** sur des modèles 7B+ — où le rang reste petit (`r = 8` typique) mais le scaling devient un hyperparamètre de calibration fin.\n", - "\n", - "**Branchement** : FT-00a (mécanisme) → FT-00b (réglage) → FT-01 (LoRA avec `peft` sur GPT-2) → FT-02 (QLoRA 4-bit) → FT-06 (vision-langage)." - ] } ], "metadata": { @@ -1556,14 +1437,14 @@ }, "papermill": { "default_parameters": {}, - "duration": 116.611506, - "end_time": "2026-09-14T02:50:38.533872", + "duration": 73.051567, + "end_time": "2026-09-14T10:41:06.318184", "environment_variables": {}, "exception": null, - "input_path": "FT-00b-LoRA-Hyperparams-from-scratch.ipynb", + "input_path": "MyIA.AI.Notebooks/GenAI/FineTuning/FT-00b-LoRA-Hyperparams-from-scratch.ipynb", "output_path": "FT-00b-LoRA-Hyperparams-from-scratch.ipynb", "parameters": {}, - "start_time": "2026-09-14T02:48:41.922366", + "start_time": "2026-09-14T10:39:53.266617", "version": "2.6.0" } },