JaminOne
/

testing_nlp

Model card Files Files and versions Community

JaminOne commited on Jun 26, 2023

Commit

7000ec4

•

1 Parent(s): a523a0a

notebook

Browse files

Files changed (1) hide show

hugging_face_shared.ipynb +572 -0

hugging_face_shared.ipynb ADDED Viewed

	@@ -0,0 +1,572 @@

+{
+ "cells": [
+  {
+   "attachments": {},
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Use it locally"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 1,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/jamin/.pyenv/versions/3.8.13/envs/bert_nlp/lib/python3.8/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",
+      "  from .autonotebook import tqdm as notebook_tqdm\n",
+      "Downloading (…)okenizer_config.json: 1.30kB [00:00, 1.73MB/s]\n",
+      "Downloading (…)olve/main/vocab.json: 798kB [00:00, 1.11MB/s]\n",
+      "Downloading (…)olve/main/merges.txt: 456kB [00:00, 920kB/s] \n",
+      "Downloading (…)/main/tokenizer.json: 1.36MB [00:01, 1.31MB/s]\n",
+      "Downloading (…)cial_tokens_map.json: 100%|██████████| 239/239 [00:00<00:00, 91.2kB/s]\n",
+      "Downloading (…)lve/main/config.json: 1.88kB [00:00, 3.10MB/s]\n",
+      "Downloading pytorch_model.bin: 100%|██████████| 499M/499M [00:17<00:00, 28.4MB/s] \n"
+     ]
+    }
+   ],
+   "source": [
+    "from transformers import AutoTokenizer, AutoModelForSequenceClassification\n",
+    "\n",
+    "tokenizer = AutoTokenizer.from_pretrained(\"cardiffnlp/tweet-topic-21-multi\")\n",
+    "\n",
+    "model = AutoModelForSequenceClassification.from_pretrained(\"cardiffnlp/tweet-topic-21-multi\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "news_&_social_concern\n",
+      "sports\n"
+     ]
+    }
+   ],
+   "source": [
+    "from transformers import AutoModelForSequenceClassification, TFAutoModelForSequenceClassification\n",
+    "from transformers import AutoTokenizer\n",
+    "import numpy as np\n",
+    "from scipy.special import expit\n",
+    "\n",
+    "class_mapping = model.config.id2label\n",
+    "\n",
+    "text = \"It is great to see athletes promoting awareness for climate change.\"\n",
+    "tokens = tokenizer(text, return_tensors='pt')\n",
+    "output = model(**tokens)\n",
+    "\n",
+    "scores = output[0][0].detach().numpy()\n",
+    "scores = expit(scores)\n",
+    "predictions = (scores >= 0.5) * 1\n",
+    "\n",
+    "# Map to classes\n",
+    "for i in range(len(predictions)):\n",
+    "  if predictions[i]:\n",
+    "    print(class_mapping[i])"
+   ]
+  },
+  {
+   "attachments": {},
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## API"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 8,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "[[{'label': 'diaries_&_daily_life', 'score': 0.752070963382721}, {'label': 'relationships', 'score': 0.6936709880828857}, {'label': 'family', 'score': 0.10573585331439972}, {'label': 'celebrity_&_pop_culture', 'score': 0.06716123223304749}, {'label': 'other_hobbies', 'score': 0.04402140900492668}, {'label': 'film_tv_&_video', 'score': 0.029309650883078575}, {'label': 'sports', 'score': 0.026606008410453796}, {'label': 'arts_&_culture', 'score': 0.017974767833948135}, {'label': 'news_&_social_concern', 'score': 0.017801295965909958}, {'label': 'music', 'score': 0.015016891993582249}, {'label': 'gaming', 'score': 0.009747783653438091}, {'label': 'fashion_&_style', 'score': 0.0088553661480546}, {'label': 'business_&_entrepreneurs', 'score': 0.008412620052695274}, {'label': 'fitness_&_health', 'score': 0.008045237511396408}, {'label': 'youth_&_student_life', 'score': 0.006527383346110582}, {'label': 'science_&_technology', 'score': 0.006279776804149151}, {'label': 'learning_&_educational', 'score': 0.005272668786346912}, {'label': 'travel_&_adventure', 'score': 0.00523344473913312}, {'label': 'food_&_dining', 'score': 0.0045149847865104675}]]\n"
+     ]
+    }
+   ],
+   "source": [
+    "import requests\n",
+    "\n",
+    "API_TOKEN = \"YOUR_API_TOKEN\"\n",
+    "\n",
+    "API_URL = \"https://api-inference.huggingface.co/models/cardiffnlp/tweet-topic-21-multi\"\n",
+    "headers = {\"Authorization\": f\"Bearer {API_TOKEN}\"}\n",
+    "\n",
+    "def query(payload):\n",
+    "\tresponse = requests.post(API_URL, headers=headers, json=payload)\n",
+    "\treturn response.json()\n",
+    "\t\n",
+    "output = query({\n",
+    "\t\"inputs\": \"I like you. I love you\",\n",
+    "})\n",
+    "print(output)"
+   ]
+  },
+  {
+   "attachments": {},
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Train Our Own Model and Upload to Hugging Face"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 2,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Found cached dataset yelp_review_full (/Users/jamin/.cache/huggingface/datasets/yelp_review_full/yelp_review_full/1.0.0/e8e18e19d7be9e75642fc66b198abadb116f73599ec89a69ba5dd8d1e57ba0bf)\n"
+     ]
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "8207b9c42a2e4b3eb0e8cd59a27fc67b",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "  0%|          | 0/2 [00:00<?, ?it/s]"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "text/plain": [
+       "{'label': 0,\n",
+       " 'text': 'My expectations for McDonalds are t rarely high. But for one to still fail so spectacularly...that takes something special!\\\\nThe cashier took my friends\\'s order, then promptly ignored me. I had to force myself in front of a cashier who opened his register to wait on the person BEHIND me. I waited over five minutes for a gigantic order that included precisely one kid\\'s meal. After watching two people who ordered after me be handed their food, I asked where mine was. The manager started yelling at the cashiers for \\\\\"serving off their orders\\\\\" when they didn\\'t have their food. But neither cashier was anywhere near those controls, and the manager was the one serving food to customers and clearing the boards.\\\\nThe manager was rude when giving me my order. She didn\\'t make sure that I had everything ON MY RECEIPT, and never even had the decency to apologize that I felt I was getting poor service.\\\\nI\\'ve eaten at various McDonalds restaurants for over 30 years. I\\'ve worked at more than one location. I expect bad days, bad moods, and the occasional mistake. But I have yet to have a decent experience at this store. It will remain a place I avoid unless someone in my party needs to avoid illness from low blood sugar. Perhaps I should go back to the racially biased service of Steak n Shake instead!'}"
+      ]
+     },
+     "execution_count": 2,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "from datasets import load_dataset\n",
+    "\n",
+    "dataset = load_dataset(\"yelp_review_full\")\n",
+    "dataset[\"train\"][100]"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 2,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "DatasetDict({\n",
+       "    train: Dataset({\n",
+       "        features: ['label', 'text'],\n",
+       "        num_rows: 650000\n",
+       "    })\n",
+       "    test: Dataset({\n",
+       "        features: ['label', 'text'],\n",
+       "        num_rows: 50000\n",
+       "    })\n",
+       "})"
+      ]
+     },
+     "execution_count": 2,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "dataset"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 3,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Loading cached processed dataset at /Users/jamin/.cache/huggingface/datasets/yelp_review_full/yelp_review_full/1.0.0/e8e18e19d7be9e75642fc66b198abadb116f73599ec89a69ba5dd8d1e57ba0bf/cache-aad1af4c7095bfa1.arrow\n",
+      "Loading cached processed dataset at /Users/jamin/.cache/huggingface/datasets/yelp_review_full/yelp_review_full/1.0.0/e8e18e19d7be9e75642fc66b198abadb116f73599ec89a69ba5dd8d1e57ba0bf/cache-29f27748f0b54d01.arrow\n"
+     ]
+    }
+   ],
+   "source": [
+    "from transformers import AutoTokenizer\n",
+    "\n",
+    "tokenizer = AutoTokenizer.from_pretrained(\"bert-base-cased\")\n",
+    "\n",
+    "\n",
+    "def tokenize_function(examples):\n",
+    "    return tokenizer(examples[\"text\"], padding=\"max_length\", truncation=True)\n",
+    "\n",
+    "\n",
+    "tokenized_datasets = dataset.map(tokenize_function, batched=True)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Loading cached shuffled indices for dataset at /Users/jamin/.cache/huggingface/datasets/yelp_review_full/yelp_review_full/1.0.0/e8e18e19d7be9e75642fc66b198abadb116f73599ec89a69ba5dd8d1e57ba0bf/cache-11a7619c6a3c070f.arrow\n",
+      "Loading cached shuffled indices for dataset at /Users/jamin/.cache/huggingface/datasets/yelp_review_full/yelp_review_full/1.0.0/e8e18e19d7be9e75642fc66b198abadb116f73599ec89a69ba5dd8d1e57ba0bf/cache-3c5c2a245be1b332.arrow\n"
+     ]
+    }
+   ],
+   "source": [
+    "small_train_dataset = tokenized_datasets[\"train\"].shuffle(seed=42).select(range(1000))\n",
+    "small_eval_dataset = tokenized_datasets[\"test\"].shuffle(seed=42).select(range(1000))"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Some weights of the model checkpoint at bert-base-cased were not used when initializing BertForSequenceClassification: ['cls.predictions.transform.dense.weight', 'cls.predictions.transform.dense.bias', 'cls.predictions.transform.LayerNorm.weight', 'cls.predictions.bias', 'cls.seq_relationship.bias', 'cls.seq_relationship.weight', 'cls.predictions.transform.LayerNorm.bias']\n",
+      "- This IS expected if you are initializing BertForSequenceClassification from the checkpoint of a model trained on another task or with another architecture (e.g. initializing a BertForSequenceClassification model from a BertForPreTraining model).\n",
+      "- This IS NOT expected if you are initializing BertForSequenceClassification from the checkpoint of a model that you expect to be exactly identical (initializing a BertForSequenceClassification model from a BertForSequenceClassification model).\n",
+      "Some weights of BertForSequenceClassification were not initialized from the model checkpoint at bert-base-cased and are newly initialized: ['classifier.bias', 'classifier.weight']\n",
+      "You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.\n"
+     ]
+    }
+   ],
+   "source": [
+    "from transformers import AutoModelForSequenceClassification\n",
+    "\n",
+    "model = AutoModelForSequenceClassification.from_pretrained(\"bert-base-cased\", num_labels=5)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 14,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from transformers import TrainingArguments, Trainer\n",
+    "\n",
+    "training_args = TrainingArguments(output_dir=\"./output\",use_mps_device=True, push_to_hub=True)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 15,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "import numpy as np\n",
+    "import evaluate\n",
+    "\n",
+    "metric = evaluate.load(\"accuracy\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 16,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "def compute_metrics(eval_pred):\n",
+    "    logits, labels = eval_pred\n",
+    "    predictions = np.argmax(logits, axis=-1)\n",
+    "    return metric.compute(predictions=predictions, references=labels)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 17,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Cloning https://huggingface.co/JaminOne/output into local empty directory.\n"
+     ]
+    }
+   ],
+   "source": [
+    "trainer = Trainer(\n",
+    "    model=model,\n",
+    "    args=training_args,\n",
+    "    train_dataset=small_train_dataset,\n",
+    "    eval_dataset=small_eval_dataset,\n",
+    "    compute_metrics=compute_metrics,\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 18,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/jamin/.pyenv/versions/3.8.13/envs/bert_nlp/lib/python3.8/site-packages/transformers/optimization.py:411: FutureWarning: This implementation of AdamW is deprecated and will be removed in a future version. Use the PyTorch implementation torch.optim.AdamW instead, or set `no_deprecation_warning=True` to disable this warning\n",
+      "  warnings.warn(\n"
+     ]
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "182076eda0db44a0a50151543c3f910c",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "  0%|          | 0/375 [00:00<?, ?it/s]"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "{'train_runtime': 520.3828, 'train_samples_per_second': 5.765, 'train_steps_per_second': 0.721, 'train_loss': 0.5104451090494792, 'epoch': 3.0}\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "TrainOutput(global_step=375, training_loss=0.5104451090494792, metrics={'train_runtime': 520.3828, 'train_samples_per_second': 5.765, 'train_steps_per_second': 0.721, 'train_loss': 0.5104451090494792, 'epoch': 3.0})"
+      ]
+     },
+     "execution_count": 18,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "trainer.train()"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 19,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "eb1761c75ce04a00a35c888c03a3edb1",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "  0%|          | 0/125 [00:00<?, ?it/s]"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "text/plain": [
+       "{'eval_loss': 1.4900177717208862,\n",
+       " 'eval_accuracy': 0.575,\n",
+       " 'eval_runtime': 50.098,\n",
+       " 'eval_samples_per_second': 19.961,\n",
+       " 'eval_steps_per_second': 2.495,\n",
+       " 'epoch': 3.0}"
+      ]
+     },
+     "execution_count": 19,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "trainer.evaluate()"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 20,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "8307540768794d629ab41e2c9c4c4528",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "Upload file pytorch_model.bin:   0%|          | 1.00/413M [00:00<?, ?B/s]"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "31b0d6fbe6f9448c84cf5bcd6fde8fda",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "Upload file runs/Jun26_16-16-56_Jamins-MBP.local/events.out.tfevents.1687753086.Jamins-MBP.local.99394.2:   0%…"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "9ba96571a2d24847a45211701e0c90ea",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "Upload file runs/Jun26_16-16-56_Jamins-MBP.local/events.out.tfevents.1687753661.Jamins-MBP.local.99394.3:   0%…"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "070b216d93864e3e9bb360c437c8edf3",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "Upload file training_args.bin:   0%|          | 1.00/3.81k [00:00<?, ?B/s]"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "To https://huggingface.co/JaminOne/output\n",
+      "   22a0170..12eaa4e  main -> main\n",
+      "\n",
+      "To https://huggingface.co/JaminOne/output\n",
+      "   12eaa4e..0bf2de2  main -> main\n",
+      "\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "'https://huggingface.co/JaminOne/output/commit/12eaa4e4155940088dea9098d47075d967361ea8'"
+      ]
+     },
+     "execution_count": 20,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "trainer.push_to_hub()"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 24,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "CommitInfo(commit_url='https://huggingface.co/JaminOne/output/commit/ed14549910b3b5016f6f6a2b92c0abc7179420fd', commit_message='Upload tokenizer', commit_description='', oid='ed14549910b3b5016f6f6a2b92c0abc7179420fd', pr_url=None, pr_revision=None, pr_num=None)"
+      ]
+     },
+     "execution_count": 24,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "tokenizer.push_to_hub(\"output\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 26,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "[[{'label': 'LABEL_4', 'score': 0.9605125784873962}, {'label': 'LABEL_3', 'score': 0.028813829645514488}, {'label': 'LABEL_0', 'score': 0.005277871619910002}, {'label': 'LABEL_1', 'score': 0.003208584152162075}, {'label': 'LABEL_2', 'score': 0.002187149366363883}]]\n"
+     ]
+    }
+   ],
+   "source": [
+    "import requests\n",
+    "\n",
+    "API_URL = \"https://api-inference.huggingface.co/models/JaminOne/output\"\n",
+    "headers = {\"Authorization\": \"Bearer YOUR_API_TOKEN\"}\n",
+    "\n",
+    "def query(payload):\n",
+    "\tresponse = requests.post(API_URL, headers=headers, json=payload)\n",
+    "\treturn response.json()\n",
+    "\t\n",
+    "output = query({\n",
+    "\t\"inputs\": \"I like you. I love you\",\n",
+    "})\n",
+    "print(output)"
+   ]
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "bert_nlp",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.8.13"
+  },
+  "orig_nbformat": 4
+ },
+ "nbformat": 4,
+ "nbformat_minor": 2
+}