Compare commits
No commits in common. "2b2cdcacf53a41f762ceae731eec077de678dad4" and "061cdb6aed1d9510e84f6c9b2c71e27f2c8c251f" have entirely different histories.
2b2cdcacf5
...
061cdb6aed
13 changed files with 243 additions and 59404 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
3
PythonAI/JupyterLab/.gitignore
vendored
3
PythonAI/JupyterLab/.gitignore
vendored
|
|
@ -1,3 +0,0 @@
|
||||||
data
|
|
||||||
model_checkpoints
|
|
||||||
logs
|
|
||||||
|
|
@ -18,7 +18,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 1,
|
"execution_count": 46,
|
||||||
"id": "178ae645-cad9-491c-9a26-c173eda34a00",
|
"id": "178ae645-cad9-491c-9a26-c173eda34a00",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
"outputs": [],
|
||||||
|
|
@ -29,7 +29,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 2,
|
"execution_count": 47,
|
||||||
"id": "540f9f69-d932-4b5a-8d70-e18a08bdcf46",
|
"id": "540f9f69-d932-4b5a-8d70-e18a08bdcf46",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -245,7 +245,7 @@
|
||||||
"[481 rows x 10 columns]"
|
"[481 rows x 10 columns]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 2,
|
"execution_count": 47,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -268,7 +268,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 3,
|
"execution_count": 48,
|
||||||
"id": "62e61e9e-f06e-4124-9a89-1979dc071c47",
|
"id": "62e61e9e-f06e-4124-9a89-1979dc071c47",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -496,7 +496,7 @@
|
||||||
"[481 rows x 11 columns]"
|
"[481 rows x 11 columns]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 3,
|
"execution_count": 48,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -521,7 +521,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 4,
|
"execution_count": 49,
|
||||||
"id": "e06e7a55-1837-456b-97ef-4c58276bd7b6",
|
"id": "e06e7a55-1837-456b-97ef-4c58276bd7b6",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -749,7 +749,7 @@
|
||||||
"[481 rows x 11 columns]"
|
"[481 rows x 11 columns]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 4,
|
"execution_count": 49,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -773,7 +773,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 5,
|
"execution_count": 50,
|
||||||
"id": "389f0c22-b210-45f3-9627-02c06d255a69",
|
"id": "389f0c22-b210-45f3-9627-02c06d255a69",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
"outputs": [],
|
||||||
|
|
@ -793,7 +793,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 6,
|
"execution_count": 51,
|
||||||
"id": "60362a3d-91ab-4666-bf26-afdeb3cd5c9c",
|
"id": "60362a3d-91ab-4666-bf26-afdeb3cd5c9c",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
"outputs": [],
|
||||||
|
|
@ -805,7 +805,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 7,
|
"execution_count": 52,
|
||||||
"id": "a68dd0eb-6085-483b-99f3-a540a35515c2",
|
"id": "a68dd0eb-6085-483b-99f3-a540a35515c2",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
"outputs": [],
|
||||||
|
|
@ -821,7 +821,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 8,
|
"execution_count": 53,
|
||||||
"id": "bb4ecdab-2136-4970-a38d-6eaec8ec92ae",
|
"id": "bb4ecdab-2136-4970-a38d-6eaec8ec92ae",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -1049,7 +1049,7 @@
|
||||||
"[481 rows x 11 columns]"
|
"[481 rows x 11 columns]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 8,
|
"execution_count": 53,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -1060,7 +1060,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 9,
|
"execution_count": 54,
|
||||||
"id": "7c8199c4-9d1b-405a-9cda-b2aef1616999",
|
"id": "7c8199c4-9d1b-405a-9cda-b2aef1616999",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -1201,7 +1201,7 @@
|
||||||
"std 0.159067 0.0 0.159067 "
|
"std 0.159067 0.0 0.159067 "
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 9,
|
"execution_count": 54,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -1212,7 +1212,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 10,
|
"execution_count": 55,
|
||||||
"id": "7468e63c-c72b-4150-bd40-6f6e61f17af8",
|
"id": "7468e63c-c72b-4150-bd40-6f6e61f17af8",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -1381,7 +1381,7 @@
|
||||||
"std 0.176697 "
|
"std 0.176697 "
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 10,
|
"execution_count": 55,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -1400,7 +1400,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": 12,
|
"execution_count": 45,
|
||||||
"id": "efa6a165-9988-420f-a2f0-10cbbcd5dc39",
|
"id": "efa6a165-9988-420f-a2f0-10cbbcd5dc39",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [
|
"outputs": [
|
||||||
|
|
@ -1442,7 +1442,7 @@
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>157.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
" <td>249.000000</td>\n",
|
" <td>249.000000</td>\n",
|
||||||
|
|
@ -1454,7 +1454,7 @@
|
||||||
" <td>16.004819</td>\n",
|
" <td>16.004819</td>\n",
|
||||||
" <td>6.371486</td>\n",
|
" <td>6.371486</td>\n",
|
||||||
" <td>8.712048</td>\n",
|
" <td>8.712048</td>\n",
|
||||||
" <td>0.230120</td>\n",
|
" <td>0.364968</td>\n",
|
||||||
" <td>4.080321</td>\n",
|
" <td>4.080321</td>\n",
|
||||||
" <td>8.360241</td>\n",
|
" <td>8.360241</td>\n",
|
||||||
" <td>0.048193</td>\n",
|
" <td>0.048193</td>\n",
|
||||||
|
|
@ -1466,7 +1466,7 @@
|
||||||
" <td>11.000615</td>\n",
|
" <td>11.000615</td>\n",
|
||||||
" <td>10.157809</td>\n",
|
" <td>10.157809</td>\n",
|
||||||
" <td>9.936468</td>\n",
|
" <td>9.936468</td>\n",
|
||||||
" <td>2.132453</td>\n",
|
" <td>2.679477</td>\n",
|
||||||
" <td>26.325536</td>\n",
|
" <td>26.325536</td>\n",
|
||||||
" <td>15.718945</td>\n",
|
" <td>15.718945</td>\n",
|
||||||
" <td>0.249368</td>\n",
|
" <td>0.249368</td>\n",
|
||||||
|
|
@ -1539,9 +1539,9 @@
|
||||||
],
|
],
|
||||||
"text/plain": [
|
"text/plain": [
|
||||||
" TMAX TMIN TOBS WESF SNOW PRCP \\\n",
|
" TMAX TMIN TOBS WESF SNOW PRCP \\\n",
|
||||||
"count 249.000000 249.000000 249.000000 249.000000 249.000000 249.000000 \n",
|
"count 249.000000 249.000000 249.000000 157.000000 249.000000 249.000000 \n",
|
||||||
"mean 16.004819 6.371486 8.712048 0.230120 4.080321 8.360241 \n",
|
"mean 16.004819 6.371486 8.712048 0.364968 4.080321 8.360241 \n",
|
||||||
"std 11.000615 10.157809 9.936468 2.132453 26.325536 15.718945 \n",
|
"std 11.000615 10.157809 9.936468 2.679477 26.325536 15.718945 \n",
|
||||||
"min -11.700000 -17.200000 -16.100000 0.000000 0.000000 0.000000 \n",
|
"min -11.700000 -17.200000 -16.100000 0.000000 0.000000 0.000000 \n",
|
||||||
"25% 6.700000 -1.700000 0.000000 0.000000 0.000000 0.000000 \n",
|
"25% 6.700000 -1.700000 0.000000 0.000000 0.000000 0.000000 \n",
|
||||||
"50% 14.400000 5.600000 8.300000 0.000000 0.000000 0.300000 \n",
|
"50% 14.400000 5.600000 8.300000 0.000000 0.000000 0.300000 \n",
|
||||||
|
|
@ -1559,7 +1559,7 @@
|
||||||
"max 2.000000 1.000000 2.000000 "
|
"max 2.000000 1.000000 2.000000 "
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"execution_count": 12,
|
"execution_count": 45,
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"output_type": "execute_result"
|
"output_type": "execute_result"
|
||||||
}
|
}
|
||||||
|
|
@ -1571,7 +1571,6 @@
|
||||||
"df_merged[\"incl_weather_true\"] = df_merged.incl_weather_true_temp.fillna(0) + df_merged.incl_weather_true_snow.fillna(0)\n",
|
"df_merged[\"incl_weather_true\"] = df_merged.incl_weather_true_temp.fillna(0) + df_merged.incl_weather_true_snow.fillna(0)\n",
|
||||||
"df_merged[\"incl_weather_false\"] = df_merged.incl_weather_false_temp.fillna(0) + df_merged.incl_weather_false_snow.fillna(0)\n",
|
"df_merged[\"incl_weather_false\"] = df_merged.incl_weather_false_temp.fillna(0) + df_merged.incl_weather_false_snow.fillna(0)\n",
|
||||||
"df_merged[\"snow\"] = df_merged.snow_temp.fillna(0) + df_merged.snow_snow.fillna(0)\n",
|
"df_merged[\"snow\"] = df_merged.snow_temp.fillna(0) + df_merged.snow_snow.fillna(0)\n",
|
||||||
"df_merged['WESF'] = df_merged['WESF'].fillna(0)\n",
|
|
||||||
"\n",
|
"\n",
|
||||||
"del df_merged[\"PRCP_temp\"]\n",
|
"del df_merged[\"PRCP_temp\"]\n",
|
||||||
"del df_merged[\"PRCP_snow\"]\n",
|
"del df_merged[\"PRCP_snow\"]\n",
|
||||||
|
|
@ -1613,7 +1612,7 @@
|
||||||
"name": "python",
|
"name": "python",
|
||||||
"nbconvert_exporter": "python",
|
"nbconvert_exporter": "python",
|
||||||
"pygments_lexer": "ipython3",
|
"pygments_lexer": "ipython3",
|
||||||
"version": "3.13.2"
|
"version": "3.12.2"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"nbformat": 4,
|
"nbformat": 4,
|
||||||
|
|
|
||||||
File diff suppressed because one or more lines are too long
|
|
@ -5,24 +5,13 @@ name = "pypi"
|
||||||
|
|
||||||
[packages]
|
[packages]
|
||||||
jupyterlab = "*"
|
jupyterlab = "*"
|
||||||
jupyterlab-vim = "*"
|
|
||||||
notebook-intelligence = "*"
|
|
||||||
pandas = "*"
|
pandas = "*"
|
||||||
numpy = "*"
|
numpy = "*"
|
||||||
scikit-learn = "*"
|
scikit-learn = "*"
|
||||||
sklearn-pandas = "*"
|
sklearn-pandas = "*"
|
||||||
matplotlib = "*"
|
matplotlib = "*"
|
||||||
tensorflow = "*"
|
|
||||||
torch = "*"
|
|
||||||
torchvision = "*"
|
|
||||||
ultralytics = "*"
|
|
||||||
statsmodels = "*"
|
|
||||||
accelerate = "*"
|
|
||||||
datasets = "*"
|
|
||||||
h2o = "*"
|
|
||||||
|
|
||||||
[dev-packages]
|
[dev-packages]
|
||||||
|
|
||||||
[requires]
|
[requires]
|
||||||
python_version = "3.12"
|
python_version = "3.13"
|
||||||
# python_version = "3.13"
|
|
||||||
|
|
|
||||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load diff
|
|
@ -1,413 +0,0 @@
|
||||||
{
|
|
||||||
"cells": [
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 21,
|
|
||||||
"id": "7018890b-b220-48a3-a1b8-d8f6c5483f6c",
|
|
||||||
"metadata": {
|
|
||||||
"editable": true,
|
|
||||||
"slideshow": {
|
|
||||||
"slide_type": ""
|
|
||||||
},
|
|
||||||
"tags": []
|
|
||||||
},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks...\n",
|
|
||||||
"To disable this warning, you can either:\n",
|
|
||||||
"\t- Avoid using `tokenizers` before the fork if possible\n",
|
|
||||||
"\t- Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"Cloning into 'gpt-neo-125M'...\n",
|
|
||||||
"remote: Enumerating objects: 65, done.\u001b[K\n",
|
|
||||||
"remote: Counting objects: 100% (5/5), done.\u001b[K\n",
|
|
||||||
"remote: Compressing objects: 100% (5/5), done.\u001b[K\n",
|
|
||||||
"remote: Total 65 (delta 1), reused 0 (delta 0), pack-reused 60 (from 1)\u001b[K\n",
|
|
||||||
"Unpacking objects: 100% (65/65), 1.11 MiB | 9.13 MiB/s, done.\n",
|
|
||||||
"Filtering content: 100% (4/4), 1.93 GiB | 54.88 MiB/s, done.\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"!git clone https://huggingface.co/EleutherAI/gpt-neo-125M"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 1,
|
|
||||||
"id": "36078c20-4185-4f45-84f2-d3e95e1dcac9",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/nvfuser-0.2.13a0+0d33366-py3.12-linux-x86_64.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/dill-0.3.9-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/lightning_utilities-0.11.8-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/opt_einsum-3.4.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/igraph-0.11.8-py3.12-linux-x86_64.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/lightning_thunder-0.2.0.dev0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/texttable-1.7.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/looseversion-1.3.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\n",
|
|
||||||
"\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m A new release of pip is available: \u001b[0m\u001b[31;49m24.3.1\u001b[0m\u001b[39;49m -> \u001b[0m\u001b[32;49m25.1.1\u001b[0m\n",
|
|
||||||
"\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m To update, run: \u001b[0m\u001b[32;49mpython -m pip install --upgrade pip\u001b[0m\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"!pip install accelerate>=0.26.0"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 6,
|
|
||||||
"id": "11d0a83e-391e-4590-9db6-6a2a3c140ae2",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks...\n",
|
|
||||||
"To disable this warning, you can either:\n",
|
|
||||||
"\t- Avoid using `tokenizers` before the fork if possible\n",
|
|
||||||
"\t- Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/nvfuser-0.2.13a0+0d33366-py3.12-linux-x86_64.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/dill-0.3.9-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/lightning_utilities-0.11.8-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/opt_einsum-3.4.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/igraph-0.11.8-py3.12-linux-x86_64.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/lightning_thunder-0.2.0.dev0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/texttable-1.7.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\u001b[33mDEPRECATION: Loading egg at /usr/local/lib/python3.12/dist-packages/looseversion-1.3.0-py3.12.egg is deprecated. pip 25.1 will enforce this behaviour change. A possible replacement is to use pip for package installation. Discussion can be found at https://github.com/pypa/pip/issues/12330\u001b[0m\u001b[33m\n",
|
|
||||||
"\u001b[0m\n",
|
|
||||||
"\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m A new release of pip is available: \u001b[0m\u001b[31;49m24.3.1\u001b[0m\u001b[39;49m -> \u001b[0m\u001b[32;49m25.1.1\u001b[0m\n",
|
|
||||||
"\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m To update, run: \u001b[0m\u001b[32;49mpython -m pip install --upgrade pip\u001b[0m\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"!pip install datasets transformers torch gradio --quiet"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 1,
|
|
||||||
"id": "7057ed8b-c98c-446e-b40b-2de8a800cb7c",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"/usr/local/lib/python3.12/dist-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",
|
|
||||||
" from .autonotebook import tqdm as notebook_tqdm\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"from transformers import AutoTokenizer, AutoModelForCausalLM\n",
|
|
||||||
"import torch\n",
|
|
||||||
"\n",
|
|
||||||
"# Ścieżka do lokalnego modelu\n",
|
|
||||||
"local_model_path = \"./gpt-neo-125M\"\n",
|
|
||||||
"\n",
|
|
||||||
"# Załaduj tokenizer i model w trybie offline\n",
|
|
||||||
"tokenizer = AutoTokenizer.from_pretrained(local_model_path, local_files_only=True)\n",
|
|
||||||
"tokenizer.pad_token = tokenizer.eos_token\n",
|
|
||||||
"model = AutoModelForCausalLM.from_pretrained(local_model_path, local_files_only=True)\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 17,
|
|
||||||
"id": "3575956e-aa32-4173-8678-9a34b838ce3c",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"Map: 100%|██████████| 20/20 [00:00<00:00, 1442.80 examples/s]"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"Dataset({\n",
|
|
||||||
" features: ['input_ids', 'attention_mask'],\n",
|
|
||||||
" num_rows: 20\n",
|
|
||||||
"})\n",
|
|
||||||
"Liczba próbek po tokenizacji: 20\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"from datasets import Dataset\n",
|
|
||||||
"from transformers import AutoTokenizer, AutoModelForCausalLM, TrainingArguments, Trainer, DataCollatorForLanguageModeling\n",
|
|
||||||
"\n",
|
|
||||||
"# Ścieżka do lokalnego modelu\n",
|
|
||||||
"local_model_path = \"./gpt-neo-125M\"\n",
|
|
||||||
"\n",
|
|
||||||
"# Tokenizer\n",
|
|
||||||
"tokenizer = AutoTokenizer.from_pretrained(local_model_path, local_files_only=True)\n",
|
|
||||||
"tokenizer.pad_token = tokenizer.eos_token # <-- wymagane do paddingu\n",
|
|
||||||
"\n",
|
|
||||||
"# Wczytanie danych z pliku\n",
|
|
||||||
"with open(\"data.txt\", encoding=\"utf-8\") as f:\n",
|
|
||||||
" lines = [line.strip() for line in f if line.strip()] # Usuwamy puste linie\n",
|
|
||||||
"\n",
|
|
||||||
"# Budujemy Dataset z listy słowników\n",
|
|
||||||
"data = [{\"text\": line} for line in lines]\n",
|
|
||||||
"raw_dataset = Dataset.from_list(data)\n",
|
|
||||||
"\n",
|
|
||||||
"# Funkcja tokenizująca\n",
|
|
||||||
"def tokenize_function(example):\n",
|
|
||||||
" return tokenizer(\n",
|
|
||||||
" example[\"text\"],\n",
|
|
||||||
" truncation=True,\n",
|
|
||||||
" max_length=128,\n",
|
|
||||||
" padding=\"max_length\"\n",
|
|
||||||
" )\n",
|
|
||||||
"\n",
|
|
||||||
"# Tokenizacja datasetu\n",
|
|
||||||
"tokenized_dataset = raw_dataset.map(tokenize_function, batched=True, remove_columns=[\"text\"])\n",
|
|
||||||
"\n",
|
|
||||||
"# Sprawdź czy dane są OK\n",
|
|
||||||
"print(tokenized_dataset)\n",
|
|
||||||
"print(f\"Liczba próbek po tokenizacji: {len(tokenized_dataset)}\")"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 18,
|
|
||||||
"id": "c5eedf83-e003-434e-af7c-a57d6efdd120",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stderr",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"/tmp/ipykernel_6455/2971221750.py:23: FutureWarning: `tokenizer` is deprecated and will be removed in version 5.0.0 for `Trainer.__init__`. Use `processing_class` instead.\n",
|
|
||||||
" trainer = Trainer(\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"data": {
|
|
||||||
"text/html": [
|
|
||||||
"\n",
|
|
||||||
" <div>\n",
|
|
||||||
" \n",
|
|
||||||
" <progress value='60' max='60' style='width:300px; height:20px; vertical-align: middle;'></progress>\n",
|
|
||||||
" [60/60 00:12, Epoch 3/3]\n",
|
|
||||||
" </div>\n",
|
|
||||||
" <table border=\"1\" class=\"dataframe\">\n",
|
|
||||||
" <thead>\n",
|
|
||||||
" <tr style=\"text-align: left;\">\n",
|
|
||||||
" <th>Step</th>\n",
|
|
||||||
" <th>Training Loss</th>\n",
|
|
||||||
" </tr>\n",
|
|
||||||
" </thead>\n",
|
|
||||||
" <tbody>\n",
|
|
||||||
" <tr>\n",
|
|
||||||
" <td>50</td>\n",
|
|
||||||
" <td>2.103000</td>\n",
|
|
||||||
" </tr>\n",
|
|
||||||
" </tbody>\n",
|
|
||||||
"</table><p>"
|
|
||||||
],
|
|
||||||
"text/plain": [
|
|
||||||
"<IPython.core.display.HTML object>"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"metadata": {},
|
|
||||||
"output_type": "display_data"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"data": {
|
|
||||||
"text/plain": [
|
|
||||||
"('./gpt-neo-finetuned/tokenizer_config.json',\n",
|
|
||||||
" './gpt-neo-finetuned/special_tokens_map.json',\n",
|
|
||||||
" './gpt-neo-finetuned/vocab.json',\n",
|
|
||||||
" './gpt-neo-finetuned/merges.txt',\n",
|
|
||||||
" './gpt-neo-finetuned/added_tokens.json',\n",
|
|
||||||
" './gpt-neo-finetuned/tokenizer.json')"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"execution_count": 18,
|
|
||||||
"metadata": {},
|
|
||||||
"output_type": "execute_result"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"# Wczytanie modelu\n",
|
|
||||||
"model = AutoModelForCausalLM.from_pretrained(local_model_path, local_files_only=True)\n",
|
|
||||||
"\n",
|
|
||||||
"# Collator (bez maskowania)\n",
|
|
||||||
"data_collator = DataCollatorForLanguageModeling(\n",
|
|
||||||
" tokenizer=tokenizer,\n",
|
|
||||||
" mlm=False\n",
|
|
||||||
")\n",
|
|
||||||
"\n",
|
|
||||||
"# Argumenty treningowe\n",
|
|
||||||
"training_args = TrainingArguments(\n",
|
|
||||||
" output_dir=\"./gpt-neo-finetuned\",\n",
|
|
||||||
" overwrite_output_dir=True,\n",
|
|
||||||
" per_device_train_batch_size=1,\n",
|
|
||||||
" num_train_epochs=3,\n",
|
|
||||||
" save_steps=500,\n",
|
|
||||||
" logging_steps=50,\n",
|
|
||||||
" prediction_loss_only=True,\n",
|
|
||||||
" fp16=True\n",
|
|
||||||
")\n",
|
|
||||||
"\n",
|
|
||||||
"# Tworzymy Trainer\n",
|
|
||||||
"trainer = Trainer(\n",
|
|
||||||
" model=model,\n",
|
|
||||||
" args=training_args,\n",
|
|
||||||
" train_dataset=tokenized_dataset, # <-- to była literówka, nie \"train_dataset\"\n",
|
|
||||||
" tokenizer=tokenizer,\n",
|
|
||||||
" data_collator=data_collator\n",
|
|
||||||
")\n",
|
|
||||||
"\n",
|
|
||||||
"# Start treningu\n",
|
|
||||||
"trainer.train()\n",
|
|
||||||
"\n",
|
|
||||||
"# Zapis modelu i tokenizer\n",
|
|
||||||
"trainer.save_model(\"./gpt-neo-finetuned\")\n",
|
|
||||||
"tokenizer.save_pretrained(\"./gpt-neo-finetuned\")\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 19,
|
|
||||||
"id": "0d5c46d3-e115-4a68-9337-1f20b62a21a4",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
|
||||||
"model_path = \"./gpt-neo-finetuned\"\n",
|
|
||||||
"\n",
|
|
||||||
"tokenizer = AutoTokenizer.from_pretrained(model_path)\n",
|
|
||||||
"model = AutoModelForCausalLM.from_pretrained(model_path)"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 20,
|
|
||||||
"id": "e656cc27-3748-4530-9ae5-05b2f8e9e105",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"* Running on local URL: http://127.0.0.1:7863\n",
|
|
||||||
"* Running on public URL: https://596aa820b4401f3637.gradio.live\n",
|
|
||||||
"\n",
|
|
||||||
"This share link expires in 1 week. For free permanent hosting and GPU upgrades, run `gradio deploy` from the terminal in the working directory to deploy to Hugging Face Spaces (https://huggingface.co/spaces)\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"data": {
|
|
||||||
"text/html": [
|
|
||||||
"<div><iframe src=\"https://596aa820b4401f3637.gradio.live\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"
|
|
||||||
],
|
|
||||||
"text/plain": [
|
|
||||||
"<IPython.core.display.HTML object>"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"metadata": {},
|
|
||||||
"output_type": "display_data"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"data": {
|
|
||||||
"text/plain": []
|
|
||||||
},
|
|
||||||
"execution_count": 20,
|
|
||||||
"metadata": {},
|
|
||||||
"output_type": "execute_result"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"import gradio as gr\n",
|
|
||||||
"\n",
|
|
||||||
"def chat(message, chat_history):\n",
|
|
||||||
" if chat_history:\n",
|
|
||||||
" prompt = chat_history + f\"\\nUser: {message}\\nAI:\"\n",
|
|
||||||
" else:\n",
|
|
||||||
" prompt = f\"User: {message}\\nAI:\"\n",
|
|
||||||
" \n",
|
|
||||||
" inputs = tokenizer(prompt, return_tensors=\"pt\")\n",
|
|
||||||
" outputs = model.generate(\n",
|
|
||||||
" **inputs,\n",
|
|
||||||
" max_length=len(inputs[\"input_ids\"][0]) + 200,\n",
|
|
||||||
" temperature=0.7,\n",
|
|
||||||
" pad_token_id=tokenizer.eos_token_id,\n",
|
|
||||||
" do_sample=True,\n",
|
|
||||||
" top_p=0.9\n",
|
|
||||||
" )\n",
|
|
||||||
" full_response = tokenizer.decode(outputs[0], skip_special_tokens=True)\n",
|
|
||||||
" response = full_response.split(\"AI:\")[-1].strip()\n",
|
|
||||||
" chat_history += f\"\\nUser: {message}\\nAI: {response}\"\n",
|
|
||||||
" return chat_history, chat_history\n",
|
|
||||||
"\n",
|
|
||||||
"# Gradio UI w trybie notebookowym\n",
|
|
||||||
"with gr.Blocks() as demo:\n",
|
|
||||||
" gr.Markdown(\"### Lokalny Czat z GPT\")\n",
|
|
||||||
" chatbot_output = gr.Textbox(label=\"Historia rozmowy\", lines=20, interactive=False)\n",
|
|
||||||
" user_input = gr.Textbox(label=\"Twoje pytanie\", placeholder=\"Zadaj pytanie i naciśnij Enter\")\n",
|
|
||||||
" state = gr.State(\"\")\n",
|
|
||||||
"\n",
|
|
||||||
" user_input.submit(chat, [user_input, state], [chatbot_output, state])\n",
|
|
||||||
" user_input.submit(lambda: \"\", None, user_input) # Czyści input po wysłaniu\n",
|
|
||||||
"\n",
|
|
||||||
"demo.launch(inline=True,share=True)"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"metadata": {
|
|
||||||
"kernelspec": {
|
|
||||||
"display_name": "Python 3 (ipykernel)",
|
|
||||||
"language": "python",
|
|
||||||
"name": "python3"
|
|
||||||
},
|
|
||||||
"language_info": {
|
|
||||||
"codemirror_mode": {
|
|
||||||
"name": "ipython",
|
|
||||||
"version": 3
|
|
||||||
},
|
|
||||||
"file_extension": ".py",
|
|
||||||
"mimetype": "text/x-python",
|
|
||||||
"name": "python",
|
|
||||||
"nbconvert_exporter": "python",
|
|
||||||
"pygments_lexer": "ipython3",
|
|
||||||
"version": "3.12.3"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"nbformat": 4,
|
|
||||||
"nbformat_minor": 5
|
|
||||||
}
|
|
||||||
File diff suppressed because it is too large
Load diff
|
|
@ -7,4 +7,4 @@ To jest lista traningów jakie odbyłem
|
||||||
### C/C++ Secure Coding - 18-20.03.2023
|
### C/C++ Secure Coding - 18-20.03.2023
|
||||||
### Zaawansowane Tehcniki Programowania C++ - 26-26.04.2024
|
### Zaawansowane Tehcniki Programowania C++ - 26-26.04.2024
|
||||||
|
|
||||||
### PythonAI 25.03-21.05.2025 7 dni w 3 turach
|
### PythonAI 25-26.03.2025
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue