diff --git a/lab-extractive-question-answering.ipynb b/lab-extractive-question-answering.ipynb index 2d7d5e3..4244128 100644 --- a/lab-extractive-question-answering.ipynb +++ b/lab-extractive-question-answering.ipynb @@ -21,7 +21,9 @@ { "cell_type": "markdown", "id": "e3f44179", - "metadata": {}, + "metadata": { + "id": "e3f44179" + }, "source": [ "
\n", "\n", @@ -88,7 +90,7 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 1, "id": "expressed-executive", "metadata": { "execution": { @@ -115,7 +117,7 @@ " pinecone==5.4.2 \\\n", " sentence-transformers==3.3.1 \\\n", " huggingface_hub==0.26.5\n", - " \n", + "\n", "!rm -rf ~/.cache/huggingface\n", "\n", "\n", @@ -134,9 +136,24 @@ "execution_count": null, "id": "2DCPtl6IhgSz", "metadata": { - "id": "2DCPtl6IhgSz" + "id": "2DCPtl6IhgSz", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "be49cbff-1d67-4920-b3c5-04e2a078212a" }, - "outputs": [], + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "⚠️ WARNING:\n", + "This code cell will restart the kernel to load the newly installed packages...\n", + "If you used 'Run all', it will stop here — that's expected.\n", + "See the note below on how to continue running the rest of the notebook.\n" + ] + } + ], "source": [ "# Force-kill the kernel process to restart it, ensuring the newly\n", "# installed/uninstalled package versions are loaded in a fresh process\n", @@ -159,7 +176,9 @@ { "cell_type": "markdown", "id": "d339487b", - "metadata": {}, + "metadata": { + "id": "d339487b" + }, "source": [ "
\n", "\n", @@ -175,14 +194,16 @@ { "cell_type": "markdown", "id": "2a62f4d8", - "metadata": {}, + "metadata": { + "id": "2a62f4d8" + }, "source": [ "# Setup for Pinecone" ] }, { "cell_type": "code", - "execution_count": null, + "execution_count": 1, "id": "WDBzNVM4hqTB", "metadata": { "id": "WDBzNVM4hqTB" @@ -195,13 +216,13 @@ "_ = load_dotenv(find_dotenv())\n", "\n", "#\n", - "# For this notebook, you'll need a Pinecone API key, \n", + "# For this notebook, you'll need a Pinecone API key,\n", "# to get one, sign up for free at https://app.pinecone.io/,\n", "# then go to \"API Keys\" and copy the default key (or create a new one).\n", "#\n", "# Once you have your Pinecone API key,\n", - "# load it from Colab's \"Secrets\" manager \n", - "# (the key icon in the left sidebar). \n", + "# load it from Colab's \"Secrets\" manager\n", + "# (the key icon in the left sidebar).\n", "# Add a secret named PINECONE_API_KEY there and enable notebook access.\n", "#\n", "PINECONE_API_KEY= userdata.get('PINECONE_API_KEY')" @@ -229,12 +250,145 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 2, "id": "J250IJeh7NIb", "metadata": { - "id": "J250IJeh7NIb" + "id": "J250IJeh7NIb", + "colab": { + "base_uri": "https://localhost:8080/", + "height": 205, + "referenced_widgets": [ + "593e5d8e8b1d40c2a8c60670cb85d7f8", + "65e66e12b91948a2aea7afae239c984a", + "53a872ceb3264fb5b0ef536da08373bd", + "d08bd24cf7694bb1a8bceb03d594fd50", + "7bbfe80c4a184882a8ed5952ecb139fd", + "e975b779605c42c19b353dae656443e2", + "9ac7e704baa64574a5581ac66bbe1c2c", + "0db9932aaea74d958c54fab427f73f95", + "dd77b2d956c9483bac6b443cfd2fe48f", + "c411c84c1acc41b0bcaae14873b7ec58", + "1fb975c9381c41de93380ebc58285f82", + "6afd7e51c8f1468fafb379396c3664bf", + "302b85be8ecc4108a79aa12503c6d224", + "5da13a9f52944bbd8520fc7e5632554e", + "a16ff559d7a043dfabcfb42f4300044f", + "12287e5296ca49a29e5f704d3e88c08c", + "60129cc7fb204fc2a50619661cf435fd", + "40bde8da8acd4e35b75bd2787e7e59c7", + "106caf326e784e71855b6cb96c22477f", + "584c592c233d41c7961ee60376a5d0aa", + "6e221ad01b874c49a51cd5a791f3cee2", + "afea77164a754c669318221024f88aa1", + "bc3a42ce10df490987ea7546fbaee93e", + "75d3f0b4a8a3441b932be65004a09677", + "cfe7343dc13742aeabe11d4dff3a8a45", + "ea4ffd63765c4f7ea0b2ba35e3cb2f4d", + "f1a75778bd1645129665f2c6bbfa4a17", + "2a846d257e4e494bb01a0f2666bf6d17", + "c2bad913f9cf46ad9750313275e4e44c", + "9694bec3703842cc853bdeb41f3180ad", + "cbb46d00e2cd41afbee8a4b8f668b591", + "157021b2a75746ddb1e8fe885c2aad7e", + "7c545dbe31824583a0872d43bf43e6a2", + "06bc8161657440109dfef961fcb5957c", + "b1d344529a0d4df1b1f20baa124781f3", + "cdf88dbbd19d42c183a480a017b19361", + "d070fb98db594f44a8d6c05293df93c2", + "96cc7d7fe9ab42908961674b62bfff0c", + "2f37a7b682e24e07aa23e4c299410644", + "8d6f82ce4f9e45f0be1f1228f7a267a2", + "9399fe4a97104b77b057fb0ae70c49f5", + "1e921e170c634bdaa69d72555ebafafe", + "74c788e7191548549cb6c3380f218423", + "c8f53d81143e400b8fcfa22b043d8f1d", + "155ac00601c04486acd17e6b5db23e95", + "569e7d15a920465483f9717e3078f20d", + "924a0828b4444cd0ace15b5f86064cb3", + "c0db58fca5e34d1b92b6150c269ec525", + "a21a73cff4314c0ea65cd7c49f43cede", + "7ad741f0723d43baac2dd72bfe69d0dd", + "3feec28c8b604742802accc796dde640", + "293289be58ef4b1cb99b2030e087335d", + "f704caf15199446b9f231abd82a5094a", + "98e97ee0f36c4e3c88e7520e3d7d09f1", + "7533d8c28cbc416cb9c9019e6be1430c" + ] + }, + "outputId": "77660f28-dcbe-4ec1-d0f5-1fd0c041fdbb" }, - "outputs": [], + "outputs": [ + { + "output_type": "display_data", + "data": { + "text/plain": [ + "README.md: 0.00B [00:00, ?B/s]" + ], + "application/vnd.jupyter.widget-view+json": { + "version_major": 2, + "version_minor": 0, + "model_id": "593e5d8e8b1d40c2a8c60670cb85d7f8" + } + }, + "metadata": {} + }, + { + "output_type": "display_data", + "data": { + "text/plain": [ + "train-00000-of-00001.parquet: 0%| | 0.00/14.5M [00:00\n", + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
titlecontext
0University_of_Notre_DameArchitecturally, the school has a Catholic cha...
5University_of_Notre_DameAs at most other universities, Notre Dame's st...
10University_of_Notre_DameThe university is the major seat of the Congre...
15University_of_Notre_DameThe College of Engineering was established in ...
20University_of_Notre_DameAll of Notre Dame's undergraduate students are...
.........
87574KathmanduInstitute of Medicine, the central college of ...
87579KathmanduFootball and Cricket are the most popular spor...
87584KathmanduThe total length of roads in Nepal is recorded...
87589KathmanduThe main international airport serving Kathman...
87594KathmanduKathmandu Metropolitan City (KMC), in order to...
\n", + "

18891 rows × 2 columns

\n", + "
\n", + "
\n", + "\n", + "
\n", + " \n", + "\n", + " \n", + "\n", + " \n", + "
\n", + "\n", + "\n", + "
\n", + " \n", + " \n", + " \n", + "
\n", + "\n", + "
\n", + "
\n" + ], + "application/vnd.google.colaboratory.intrinsic+json": { + "type": "dataframe", + "variable_name": "df", + "summary": "{\n \"name\": \"df\",\n \"rows\": 18891,\n \"fields\": [\n {\n \"column\": \"title\",\n \"properties\": {\n \"dtype\": \"category\",\n \"num_unique_values\": 442,\n \"samples\": [\n \"Dominican_Order\",\n \"Brigham_Young_University\",\n \"Matter\"\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n },\n {\n \"column\": \"context\",\n \"properties\": {\n \"dtype\": \"string\",\n \"num_unique_values\": 18891,\n \"samples\": [\n \"More commonly, in cases where there are three or more parties, no one party is likely to gain power alone, and parties work with each other to form coalition governments. This has been an emerging trend in the politics of the Republic of Ireland since the 1980s and is almost always the case in Germany on national and state level, and in most constituencies at the communal level. Furthermore, since the forming of the Republic of Iceland there has never been a government not led by a coalition (usually of the Independence Party and one other (often the Social Democratic Alliance). A similar situation exists in the Republic of Ireland; since 1989, no one party has held power on its own. Since then, numerous coalition governments have been formed. These coalitions have been exclusively led by one of either Fianna F\\u00e1il or Fine Gael. Political change is often easier with a coalition government than in one-party or two-party dominant systems.[dubious \\u2013 discuss] If factions in a two-party system are in fundamental disagreement on policy goals, or even principles, they can be slow to make policy changes, which appears to be the case now in the U.S. with power split between Democrats and Republicans. Still coalition governments struggle, sometimes for years, to change policy and often fail altogether, post World War II France and Italy being prime examples. When one party in a two-party system controls all elective branches, however, policy changes can be both swift and significant. Democrats Woodrow Wilson, Franklin Roosevelt and Lyndon Johnson were beneficiaries of such fortuitous circumstances, as were Republicans as far removed in time as Abraham Lincoln and Ronald Reagan. Barack Obama briefly had such an advantage between 2009 and 2011.\",\n \"There has been some concern over the potential adverse environmental and ecosystem effects caused by the influx of visitors. Some environmentalists and scientists have made a call for stricter regulations for ships and a tourism quota. The primary response by Antarctic Treaty Parties has been to develop, through their Committee for Environmental Protection and in partnership with IAATO, \\\"site use guidelines\\\" setting landing limits and closed or restricted zones on the more frequently visited sites. Antarctic sightseeing flights (which did not land) operated out of Australia and New Zealand until the fatal crash of Air New Zealand Flight 901 in 1979 on Mount Erebus, which killed all 257 aboard. Qantas resumed commercial overflights to Antarctica from Australia in the mid-1990s.\",\n \"After World War II, the Guam Organic Act of 1950 established Guam as an unincorporated organized territory of the United States, provided for the structure of the island's civilian government, and granted the people U.S. citizenship. The Governor of Guam was federally appointed until 1968, when the Guam Elective Governor Act provided for the office's popular election.:242 Since Guam is not a U.S. state, U.S. citizens residing on Guam are not allowed to vote for president and their congressional representative is a non-voting member.\"\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n }\n ]\n}" + } + }, + "metadata": {}, + "execution_count": 11 + } + ], "source": [ "# select only title and context column\n", - "df = None\n", + "# Check if df exists and is not None; if not, load the dataset.\n", + "if 'df' not in globals() or df is None:\n", + " print(\"Warning: 'df' is not defined or is None. Attempting to load the dataset.\")\n", + " from datasets import load_dataset\n", + " df = load_dataset(\"rajpurkar/squad\", split=\"train\").to_pandas()\n", + "\n", + "df = df[['title', 'context']]\n", "# drop rows containing duplicate context passages\n", - "df = None\n", + "df = df.drop_duplicates(subset=['context'])\n", "df" ] }, @@ -280,7 +717,7 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 12, "id": "092d1e71", "metadata": { "id": "092d1e71" @@ -311,21 +748,26 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 13, "id": "b3206184", "metadata": { "id": "b3206184" }, "outputs": [], "source": [ - "index_name = None\n", + "index_name = \"question-answering\"\n", "\n", "# check if the extractive-question-answering index exists\n", - "if index_name not in pinecone.list_indexes().names():\n", + "if index_name not in pc.list_indexes().names():\n", " # create the index if it does not exist\n", - " None\n", + " pc.create_index(\n", + " name=index_name,\n", + " dimension=384,\n", + " metric=\"cosine\",\n", + " spec=spec\n", + " )\n", "# connect to extractive-question-answering index we created\n", - "index = pinecone.Index(index_name)" + "index = pc.Index(index_name)" ] }, { @@ -357,7 +799,7 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 14, "id": "31a85bb3", "metadata": { "id": "31a85bb3" @@ -397,37 +839,452 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 15, "id": "a17824ef", "metadata": { "id": "a17824ef", - "tags": [] + "tags": [], + "colab": { + "base_uri": "https://localhost:8080/", + "height": 265, + "referenced_widgets": [ + "a4f819cc533d46038a3fda028142eb6c", + "3e12c04704324a4aa473d3f7cf4ca72e", + "8285eda9ee1a48f99c503eff8861c47f", + "3d8adfa2ef9a4754b8ad715abcde55ac", + "88200f3537d341e2b7c1f4a33e503671", + "1e44a5d99a124cab8a49e35863a3efc4", + "599cc72538e048f1810ae7f8693a8a1e", + "ee834c0af93a4391b4ebb2a833b86949", + "0a85affb85cb4fdd8aec6891c767fe70", + "f17c870580a246248caf8e4ae2acd6f0", + "b2ab4b4d9b77431abbd22f9f838697fb" + ] + }, + "outputId": "6c6f8f26-eb22-4da9-ea7d-76734d0799da" + }, + "outputs": [ + { + "output_type": "display_data", + "data": { + "text/plain": [ + " 0%| | 0/296 [00:00\u001b[0;34m()\u001b[0m\n\u001b[1;32m 10\u001b[0m \u001b[0mbatch\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mdf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0miloc\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mi\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0mi_end\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 11\u001b[0m \u001b[0;31m# generate embeddings for batch\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 12\u001b[0;31m \u001b[0memb\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mretriever\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mencode\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mbatch\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'context'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mtolist\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mtolist\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 13\u001b[0m \u001b[0;31m# get metadata\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 14\u001b[0m \u001b[0mmeta\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;34m[\u001b[0m\u001b[0;34m{\u001b[0m\u001b[0;34m'title'\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mr\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'title'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m'context'\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mr\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'context'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m}\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0midx\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mr\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mbatch\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0miterrows\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;31mAttributeError\u001b[0m: 'NoneType' object has no attribute 'encode'" + ] + } + ], + "source": [ + "from tqdm.auto import tqdm\n", + "\n", + "# we will use batches of 64\n", + "batch_size = 64\n", + "\n", + "for i in tqdm(range(0, len(df), batch_size)):\n", + " # find end of batch\n", + " i_end = min(i + batch_size, len(df))\n", + " # extract batch\n", + " batch = df.iloc[i:i_end]\n", + " # generate embeddings for batch\n", + " emb = retriever.encode(batch['context'].tolist()).tolist()\n", + " # get metadata\n", + " meta = [{'title': r['title'], 'context': r['context']} for idx, r in batch.iterrows()]\n", + " # create unique IDs\n", + " ids = [f\"{idx}\" for idx in range(i, i_end)]\n", + " # add all to upsert list\n", + " to_upsert = list(zip(ids, emb, meta))\n", + " # upsert/insert these records to pinecone\n", + " _ = index.upsert(vectors=to_upsert)\n", + "\n", + "# check that we have all vectors in index\n", + "index.describe_index_stats()" + ] + }, + { + "cell_type": "code", + "metadata": { + "colab": { + "base_uri": "https://localhost:8080/", + "height": 490, + "referenced_widgets": [ + "e14ef927adc64b479762e920eea55cce", + "6df29a7e4b8b4321aaab4b924201ca7e", + "d6eb9d8d20594320959f8ec891960cc1", + "7f49869d72c5456bb24c86234dec07ee", + "4d9cb604b4d64969b56420fd9162f179", + "6d7d2836f9614147b2f4b81478b4cf08", + "1cb6a2dc7a024f7995ae376e4d358196", + "a34e9a8b36634497bc52d968ad4dcdbe", + "382d405b90354aeb9307915b718bdada", + "e08d12da6b384419bfc121d5529225d3", + "319c8a462dd049548045497dcdffe3f5", + "93154b3eaf664addb14830122485f811", + "a4d193e041254a02b81bcbb8a29e9748", + "0951c42ac6d74239a3664eeac67f4e84", + "006546d6061a471c95ac008d79d07c88", + "416076a5f7e648f3bfdf5bc6c6096e9d", + "0510c2af71124debb527755f0c229d6e", + "eca35c99d24e4f1dbe9ace77eb535d02", + "6a0e42978cf441508204dfa6ec1b11e4", + "a0be0f79cc2f482a8b3e7cabf9fba5b5", + "ac8e6c8a94e94f3d91086af1d3cc2245", + "43deb5625840474ca0acb41d4324dc8e", + "41a429fe518b46febc1c5ab15ce766cb", + "624e5c87dd174c618a89541ba3fb6d55", + "a358484318d645f28da500c7aa5148f0", + "e7a24c8791dd40898e71bfb669bdc93e", + "771811d2adab4c6191cc1eccc457fcba", + "78ba305830fe423d8bee38ac2033d908", + "ecda22b6725645909d8ecd4863c3b308", + "eb948c547558434d956b8417822129b3", + "5feb821c30ac4ddeb5da445668d49f8e", + "e42fff21e8d246adaa2a97f8907761bc", + "7de0437445624c7c8f8f5b8b6433dc61", + "7dad152587d44ee7a7944b10635cc3b5", + "f04e539866dc470aa792a67dae035875", + "66e3087f26bd4f52ab89240bb3f60d94", + "c312f19b54ec449e94d7cb18cee0cd86", + "9aa4dd37aeb64ee6a8a80ca0633eb588", + "34ee956bd27b4f4d8846813f60d19475", + "77738349b3c5451388827375027e2c4e", + "0b1013e53ef742ada8c170e7dd284115", + "f4af6480bea94c2a89369603f52c2e8a", + "ae5d9628acc54c9595d8fe30968e58e0", + "c9f3e8efc2dd46fa86a9676cf8999281", + "ac9c4a5463724ae6a4d83e0056ac21cb", + "d2661a15896c4a869d81978fe8821e5a", + "ad0f67082cc6467b8d237859b847bf3f", + "d8508c117b6742218d6ec2674b68b059", + "8276beb16f0441069d15fa570f0ce447", + "d934ed5b4e16475d8d7723915637c084", + "926d7484a4ab41338ebf3267859efea1", + "b5e684b3132240c9bbc7e1e0e5f13a47", + "175a860ba30a4ba6afd052462498108f", + "37fd8059c5874d29b4c9658dd7ea0773", + "0c9c8db2bfc84aae94c94998606fa7d7", + "b66ed9a214cb4128800cb0b5f949b8d8", + "5149827c7b654670afba287812821841", + "8018a55c663640ab8426c9b362f9250b", + "0419add605fb49ef8735f348803e8467", + "cfe128d59a2240989bebe3ef9629f9de", + "d31c77bb715e47d494d860a13cbce53a", + "e435d0596b8e4893bcca7884609a26fb", + "80bad4002d1b4909aa253319d26372c7", + "54e128ac6f8542498f0bd7796cfd819e", + "a4ec47a3804848d4af9d5bc1fd616289", + "030b0ad10c83469ba25df6cbc4878e54", + "0e9c176c701d4c138beebcc8fd758af9", + "37f3bdc63e06454496be082cc6920e83", + "458c25906e6a4d2091725d3af1d015ef", + "d959f4dec95b44d1b1a46723e7db82e4", + "9e1f9496640449b0b89f9c3e49a35b52", + "d11f5536bc4143c0b9ee4aae49042c3e", + "93e1040d726140b68db07fe879215b91", + "ab72678d426b4ca1bd2f526bae40cd70", + "33b746434901455aa42532e3eace2a72", + "afb5be68eab64e33bf6058d20c5137e0", + "808d3201767444e8b80add3d345f4b7d", + "f518c44b12b3473e9070170dbb7a19ce", + "d93309b9abfb40a7b38a18311c4d0507", + "e775c0bbb3324254a809702f72e2980e", + "53076858420048df8eb3201b3e2961fb", + "774ee9be50b8426dad643a8507014904", + "d415edf6af2a4e60965f1ba3e7a2456e", + "b6b9c443a60847b1a6020526a8366d80", + "0db53258204e4cf5983abea7e56da4cb", + "5c4bedaac7e44f56bc831bf09923dd3b", + "1b7e1fc42983481581766f4f77bd6796", + "75f61b9641724c23953acb4cd65bf717", + "96c90104ed964c0086b5af9708049608", + "57b4415c63724a1a8fe0803c411e65a6", + "e70e81cf6b0645f381da8bea097d863f", + "e73b9cb2e0cc493e93553e02f224eaba", + "327655de6d734677bcd5d81b1728402f", + "eba1e30224354d9f8b0aaa84377e3df1", + "a3f335052de647c8b58dfd711f5a9610", + "856daa3ba3c04e57bece5e44dad07291", + "2af23fdf09bf4c1ca44750853264d527", + "c7eafe35777e42d9aec8e616e6da67b1", + "26c3767d793f4e5fb218b47546d14701", + "b0b359a258034c39895ba2d79b251b96", + "1ad4502db4214a02bbf61f4ff82951e9", + "cf70798b8f1e4bd2bc613075ccc66048", + "d0ddad1ddd344a2080464693c6512bdb", + "203d7a1cb1e1434e9086030f5eb50a00", + "a13ccfa8d5d144eb9e2087c3438c5559", + "2cc7526f84db437488270526707956f9", + "9ed958feaba0473fbe71cffbd4476698", + "b07b12aa24e54626951b3142171b6e4e", + "ac37aa5ef3fd46098f43fc24a303c45a", + "bbd15490274448138d9898a2e25c0f93", + "1beeb57e39d54333895024bf20d6d4c4", + "394ef1a8428240a8a9cff485a04df0cc", + "4131752b15b54a308f1902371917ec6e", + "094bb86e84c34d71801fdda0d5c99809", + "d327be6c060e44a29d20e6925bc97d55", + "5c0eba8f171a4fb986896c3c6bf53e26", + "578f57df995a43e29eea3d1377bef073", + "9f8bec1b52a24b79b5fa31a5bb26d867", + "8bf6fc390e794855ae004395d25026c0", + "f9b8c53458304d06b7c61e5d30ca3b1a", + "9c4f58d495a14635a60f0f0901ca8c95", + "d6c06c60a5094200b03df32c4d302b84", + "de1881521428471787721c7b28db8ae1", + "5c8ed8fbf08b4482929bf94f77ac81f4", + "dd7d6c4770d04afc8d3ea382d7edf678", + "691ac160b6b546b19b2aa63124ea663c", + "390defac1823469a9e168838afd3e33a", + "8c6be8ce770c4920b5d5ab7cc2537892", + "abe3bd5cdcf142c6bfb9e63b51c610f7", + "2c51e2d52f844d0eb5ca0bbfcee1c955", + "48094e516b21404bac71588948b7e239", + "220eedcbf86746b7bbf61506b3a45ec6" + ] + }, + "id": "a17842ef", + "outputId": "7fb20017-7ead-4594-9e40-012211159129" }, - "outputs": [], "source": [ "from tqdm.auto import tqdm\n", + "import torch\n", + "from sentence_transformers import SentenceTransformer\n", + "\n", + "# Check if retriever is initialized, if not, initialize it.\n", + "# This is a workaround because `retriever` was set to `None` in a previous cell.\n", + "if 'retriever' not in globals() or retriever is None or not hasattr(retriever, 'encode'):\n", + " print(\"Warning: 'retriever' not properly initialized. Initializing it now in this cell.\")\n", + " device = 'cuda' if torch.cuda.is_available() else 'cpu'\n", + " retriever = SentenceTransformer('multi-qa-MiniLM-L6-cos-v1', device=device)\n", "\n", "# we will use batches of 64\n", "batch_size = 64\n", "\n", "for i in tqdm(range(0, len(df), batch_size)):\n", " # find end of batch\n", - " None\n", + " i_end = min(i + batch_size, len(df))\n", " # extract batch\n", - " None\n", + " batch = df.iloc[i:i_end]\n", " # generate embeddings for batch\n", - " emb = None\n", + " emb = retriever.encode(batch['context'].tolist()).tolist()\n", " # get metadata\n", - " meta = None\n", + " meta = [{'title': r['title'], 'context': r['context']} for idx, r in batch.iterrows()]\n", " # create unique IDs\n", - " ids = None\n", + " ids = [f\"{idx}\" for idx in range(i, i_end)]\n", " # add all to upsert list\n", - " to_upsert = None\n", + " to_upsert = list(zip(ids, emb, meta))\n", " # upsert/insert these records to pinecone\n", " _ = index.upsert(vectors=to_upsert)\n", "\n", "# check that we have all vectors in index\n", "index.describe_index_stats()" + ], + "id": "a17842ef", + "execution_count": 16, + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "Warning: 'retriever' not properly initialized. Initializing it now in this cell.\n" + ] + }, + { + "output_type": "display_data", + "data": { + "text/plain": [ + "modules.json: 0%| | 0.00/349 [00:00" + ] + }, + "metadata": {}, + "execution_count": 17 + } + ], "source": [ "from transformers import pipeline\n", "\n", @@ -482,7 +1482,7 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 19, "id": "lyYaY3QEQiHZ", "metadata": { "id": "lyYaY3QEQiHZ" @@ -492,18 +1492,17 @@ "# gets context passages from the pinecone index\n", "def get_context(question, top_k):\n", " # generate embeddings for the question\n", - " xq = None\n", + " xq = retriever.encode(question).tolist()\n", " # search pinecone index for context passage with the answer\n", - " xc = None\n", + " xc = index.query(vector=xq, top_k=top_k, include_metadata=True)\n", " # extract the context passage from pinecone search result\n", - " c = None\n", - " return c\n", - "\n" + " c = [match['metadata']['context'] for match in xc['matches']]\n", + " return c" ] }, { "cell_type": "code", - "execution_count": null, + "execution_count": 35, "id": "Dc9VYOiUQA7B", "metadata": { "id": "Dc9VYOiUQA7B" @@ -522,18 +1521,37 @@ " answer[\"context\"] = c\n", " results.append(answer)\n", " # sort the result based on the score from reader model\n", - " sorted_result = pprint(sorted(results, key=lambda x: x['score'], reverse=True))\n", - " return sorted_result" + " sorted_results = sorted(results, key=lambda x: x['score'], reverse=True)\n", + " # The line below was previously causing the TypeError because pprint returns None.\n", + " # It's kept commented out. If you wish to print the full sorted results for debugging,\n", + " # you can uncomment it, but make sure the 'return sorted_results' line is still present.\n", + " # pprint(sorted_results)\n", + " return sorted_results" ] }, { "cell_type": "code", - "execution_count": null, + "execution_count": 21, "id": "5E3a3dkJ5ZQD", "metadata": { - "id": "5E3a3dkJ5ZQD" + "id": "5E3a3dkJ5ZQD", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "bf52ed28-aea4-4ce0-bcd9-257f8b8f7418" }, - "outputs": [], + "outputs": [ + { + "output_type": "execute_result", + "data": { + "text/plain": [ + "['Egypt was producing 691,000 bbl/d of oil and 2,141.05 Tcf of natural gas (in 2013), which makes Egypt as the largest oil producer not member of the Organization of the Petroleum Exporting Countries (OPEC) and the second-largest dry natural gas producer in Africa. In 2013, Egypt was the largest consumer of oil and natural gas in Africa, as more than 20% of total oil consumption and more than 40% of total dry natural gas consumption in Africa. Also, Egypt possesses the largest oil refinery capacity in Africa 726,000 bbl/d (in 2012). Egypt is currently planning to build its first nuclear power plant in El Dabaa city, northern Egypt.']" + ] + }, + "metadata": {}, + "execution_count": 21 + } + ], "source": [ "question = \"How much oil is Egypt producing in a day?\"\n", "context = get_context(question, top_k = 1)\n", @@ -552,12 +1570,38 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 22, "id": "DQ4GWdbMSjPl", "metadata": { - "id": "DQ4GWdbMSjPl" + "id": "DQ4GWdbMSjPl", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "7bf0b067-bd06-4477-b316-ed45ecd405a1" }, - "outputs": [], + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "[{'answer': '691,000 bbl/d',\n", + " 'context': 'Egypt was producing 691,000 bbl/d of oil and 2,141.05 Tcf of '\n", + " 'natural gas (in 2013), which makes Egypt as the largest oil '\n", + " 'producer not member of the Organization of the Petroleum '\n", + " 'Exporting Countries (OPEC) and the second-largest dry natural '\n", + " 'gas producer in Africa. In 2013, Egypt was the largest consumer '\n", + " 'of oil and natural gas in Africa, as more than 20% of total oil '\n", + " 'consumption and more than 40% of total dry natural gas '\n", + " 'consumption in Africa. Also, Egypt possesses the largest oil '\n", + " 'refinery capacity in Africa 726,000 bbl/d (in 2012). Egypt is '\n", + " 'currently planning to build its first nuclear power plant in El '\n", + " 'Dabaa city, northern Egypt.',\n", + " 'end': 33,\n", + " 'score': 0.9999852180480957,\n", + " 'start': 20}]\n" + ] + } + ], "source": [ "extract_answer(question, context)" ] @@ -574,12 +1618,36 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 23, "id": "_4NRgV4mGWoj", "metadata": { - "id": "_4NRgV4mGWoj" + "id": "_4NRgV4mGWoj", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "d9dff848-501e-41e9-b262-7157a0d2a86f" }, - "outputs": [], + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "[{'answer': 'Hurley and Chen',\n", + " 'context': 'According to a story that has often been repeated in the media, '\n", + " 'Hurley and Chen developed the idea for YouTube during the early '\n", + " 'months of 2005, after they had experienced difficulty sharing '\n", + " \"videos that had been shot at a dinner party at Chen's apartment \"\n", + " 'in San Francisco. Karim did not attend the party and denied that '\n", + " 'it had occurred, but Chen commented that the idea that YouTube '\n", + " 'was founded after a dinner party \"was probably very strengthened '\n", + " 'by marketing ideas around creating a story that was very '\n", + " 'digestible\".',\n", + " 'end': 79,\n", + " 'score': 0.9999276399612427,\n", + " 'start': 64}]\n" + ] + } + ], "source": [ "question = \"What are the first names of the men that invented youtube?\"\n", "context = get_context(question, top_k=1)\n", @@ -588,12 +1656,36 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 24, "id": "juXlctWgJgMF", "metadata": { - "id": "juXlctWgJgMF" + "id": "juXlctWgJgMF", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "28ff4bf4-57b4-41eb-f9c0-0b852f9a8d87" }, - "outputs": [], + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "[{'answer': 'his theories of special relativity and general relativity',\n", + " 'context': 'Albert Einstein is known for his theories of special relativity '\n", + " 'and general relativity. He also made important contributions to '\n", + " 'statistical mechanics, especially his mathematical treatment of '\n", + " 'Brownian motion, his resolution of the paradox of specific '\n", + " 'heats, and his connection of fluctuations and dissipation. '\n", + " 'Despite his reservations about its interpretation, Einstein also '\n", + " 'made contributions to quantum mechanics and, indirectly, quantum '\n", + " 'field theory, primarily through his theoretical studies of the '\n", + " 'photon.',\n", + " 'end': 86,\n", + " 'score': 0.9500371217727661,\n", + " 'start': 29}]\n" + ] + } + ], "source": [ "question = \"What is Albert Eistein famous for?\"\n", "context = get_context(question, top_k=1)\n", @@ -612,12 +1704,69 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 25, "id": "iXACn71xmett", "metadata": { - "id": "iXACn71xmett" + "id": "iXACn71xmett", + "colab": { + "base_uri": "https://localhost:8080/" + }, + "outputId": "9e2e0e0d-d843-430a-8fd7-f3a86b596d05" }, - "outputs": [], + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "[{'answer': 'Armstrong',\n", + " 'context': 'The trip to the Moon took just over three days. After achieving '\n", + " 'orbit, Armstrong and Aldrin transferred into the Lunar Module, '\n", + " 'named Eagle, and after a landing gear inspection by Collins '\n", + " 'remaining in the Command/Service Module Columbia, began their '\n", + " 'descent. After overcoming several computer overload alarms '\n", + " 'caused by an antenna switch left in the wrong position, and a '\n", + " 'slight downrange error, Armstrong took over manual flight '\n", + " 'control at about 180 meters (590 ft), and guided the Lunar '\n", + " 'Module to a safe landing spot at 20:18:04 UTC, July 20, 1969 '\n", + " '(3:17:04 pm CDT). The first humans on the Moon would wait '\n", + " 'another six hours before they ventured out of their craft. At '\n", + " '02:56 UTC, July 21 (9:56 pm CDT July 20), Armstrong became the '\n", + " 'first human to set foot on the Moon.',\n", + " 'end': 80,\n", + " 'score': 0.9998037815093994,\n", + " 'start': 71},\n", + " {'answer': 'Aldrin',\n", + " 'context': 'The first step was witnessed by at least one-fifth of the '\n", + " 'population of Earth, or about 723 million people. His first '\n", + " \"words when he stepped off the LM's landing footpad were, \"\n", + " '\"That\\'s one small step for [a] man, one giant leap for '\n", + " 'mankind.\" Aldrin joined him on the surface almost 20 minutes '\n", + " 'later. Altogether, they spent just under two and one-quarter '\n", + " 'hours outside their craft. The next day, they performed the '\n", + " 'first launch from another celestial body, and rendezvoused back '\n", + " 'with Columbia.',\n", + " 'end': 246,\n", + " 'score': 0.695867121219635,\n", + " 'start': 240},\n", + " {'answer': 'Frank Borman',\n", + " 'context': 'On December 21, 1968, Frank Borman, James Lovell, and William '\n", + " 'Anders became the first humans to ride the Saturn V rocket into '\n", + " 'space on Apollo 8. They also became the first to leave low-Earth '\n", + " 'orbit and go to another celestial body, and entered lunar orbit '\n", + " 'on December 24. They made ten orbits in twenty hours, and '\n", + " 'transmitted one of the most watched TV broadcasts in history, '\n", + " 'with their Christmas Eve program from lunar orbit, that '\n", + " 'concluded with a reading from the biblical Book of Genesis. Two '\n", + " 'and a half hours after the broadcast, they fired their engine to '\n", + " 'perform the first trans-Earth injection to leave lunar orbit and '\n", + " 'return to the Earth. Apollo 8 safely landed in the Pacific ocean '\n", + " \"on December 27, in NASA's first dawn splashdown and recovery.\",\n", + " 'end': 34,\n", + " 'score': 0.49246710538864136,\n", + " 'start': 22}]\n" + ] + } + ], "source": [ "question = \"Who was the first person to step foot on the moon?\"\n", "context = get_context(question, top_k=3)\n", @@ -636,7 +1785,7 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 26, "id": "d83f9f55-b099-4280-a12e-1d0192f4f5aa", "metadata": { "id": "d83f9f55-b099-4280-a12e-1d0192f4f5aa" @@ -646,6 +1795,129 @@ "pc.delete_index(index_name)" ] }, + { + "cell_type": "code", + "metadata": { + "id": "9492ef3a" + }, + "source": [ + "index_name = \"question-answering\"\n", + "\n", + "# check if the extractive-question-answering index exists\n", + "if index_name not in pc.list_indexes().names():\n", + " # create the index if it does not exist\n", + " pc.create_index(\n", + " name=index_name,\n", + " dimension=384,\n", + " metric=\"cosine\",\n", + " spec=spec\n", + " )\n", + "# connect to extractive-question-answering index we created\n", + "index = pc.Index(index_name)" + ], + "id": "9492ef3a", + "execution_count": 28, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "56452c70" + }, + "source": [ + "Now that the index is re-initialized, you need to re-generate and upsert all the embeddings for your context passages. This might take some time depending on the size of your dataset." + ], + "id": "56452c70" + }, + { + "cell_type": "code", + "metadata": { + "colab": { + "base_uri": "https://localhost:8080/", + "height": 120, + "referenced_widgets": [ + "58bc6252a9654d92a9c399d2341b5ff2", + "d5ac0cb1e66c4b76a38dd9d23c8db91d", + "ddaa8c3614504306ba13ee694a7ebd83", + "79f8635837434551814ba61c18ce8dd3", + "9cd4550b453248939030d94d509810f8", + "5b2428a4dbc3426f8baf61dfd930f850", + "324e9a91af0b4019a63c48072578c725", + "6992c739500840c08e058ea09e3fea3d", + "3365c4cdb56d46478b9c2dbd11fad224", + "6cb807d4019f454e9f7c6f36f18b0eb0", + "a44754cbcb2b434c9c40d89d85695d56" + ] + }, + "id": "8c69c215", + "outputId": "c0088999-e6b9-491b-fde2-eed9a0611f12" + }, + "source": [ + "from tqdm.auto import tqdm\n", + "import torch\n", + "from sentence_transformers import SentenceTransformer\n", + "\n", + "# Check if retriever is initialized, if not, initialize it.\n", + "# This is a workaround because `retriever` was set to `None` in a previous cell.\n", + "if 'retriever' not in globals() or retriever is None or not hasattr(retriever, 'encode'):\n", + " print(\"Warning: 'retriever' not properly initialized. Initializing it now in this cell.\")\n", + " device = 'cuda' if torch.cuda.is_available() else 'cpu'\n", + " retriever = SentenceTransformer('multi-qa-MiniLM-L6-cos-v1', device=device)\n", + "\n", + "# we will use batches of 64\n", + "batch_size = 64\n", + "\n", + "for i in tqdm(range(0, len(df), batch_size)):\n", + " # find end of batch\n", + " i_end = min(i + batch_size, len(df))\n", + " # extract batch\n", + " batch = df.iloc[i:i_end]\n", + " # generate embeddings for batch\n", + " emb = retriever.encode(batch['context'].tolist()).tolist()\n", + " # get metadata\n", + " meta = [{'title': r['title'], 'context': r['context']} for idx, r in batch.iterrows()]\n", + " # create unique IDs\n", + " ids = [f\"{idx}\" for idx in range(i, i_end)]\n", + " # add all to upsert list\n", + " to_upsert = list(zip(ids, emb, meta))\n", + " # upsert/insert these records to pinecone\n", + " _ = index.upsert(vectors=to_upsert)\n", + "\n", + "# check that we have all vectors in index\n", + "index.describe_index_stats()" + ], + "id": "8c69c215", + "execution_count": 29, + "outputs": [ + { + "output_type": "display_data", + "data": { + "text/plain": [ + " 0%| | 0/296 [00:00