devTamale2912 commited on
Commit
dbeb6e8
·
verified ·
1 Parent(s): 00ccf30

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -33,4 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
36
  chroma_db/chroma.sqlite3 filter=lfs diff=lfs merge=lfs -text
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ JM_QualitativeEvaluationOutput.pdf filter=lfs diff=lfs merge=lfs -text
37
+ JMartinez_OuraAppAssistantReport.pdf filter=lfs diff=lfs merge=lfs -text
38
  chroma_db/chroma.sqlite3 filter=lfs diff=lfs merge=lfs -text
JM_QualitativeEvaluationOutput.pdf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad4a8d27f639046318b7260e26af6e5ff9760cc68ebebcf33cc575285b95d839
3
+ size 102095
JMartinez_OuraAppAssistantReport.pdf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4e474187cc1a133ac327c587adddff3e9f0fec14bb687ddf9fbf29f54687e57
3
+ size 123984
README.md CHANGED
@@ -11,3 +11,20 @@ short_description: Specialized LLM for Oura Ring App
11
  ---
12
 
13
  Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
  ---
12
 
13
  Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
14
+
15
+ Reports:
16
+ - JMartinez_OuraAppAssistantReport.pdf
17
+ - JM_QualitativeEvaluationOutput.pdf
18
+
19
+ Data preparation scripts:
20
+ - extractOuraArticleDataChunks.ipynb
21
+ - extractOuraReddit.ipynb
22
+
23
+ Knowledge base construction:
24
+ - buildChromaDB.ipynb
25
+
26
+ Run app assistant notebook:
27
+ - runOuraAppAssistant.ipynb
28
+
29
+ Gradio deployment:
30
+ - app.py
buildChromaDB.ipynb ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "metadata": {},
6
+ "source": [
7
+ "# Load Chunks"
8
+ ]
9
+ },
10
+ {
11
+ "cell_type": "code",
12
+ "execution_count": null,
13
+ "metadata": {},
14
+ "outputs": [],
15
+ "source": [
16
+ "import json\n",
17
+ "\n",
18
+ "def load_json(file_path):\n",
19
+ " with open(file_path, 'r') as file:\n",
20
+ " data = json.load(file)\n",
21
+ " return data\n",
22
+ "\n",
23
+ "file_path = 'oura_article_data_chunked_final.json'\n",
24
+ "chunks = load_json(file_path)\n",
25
+ "\n"
26
+ ]
27
+ },
28
+ {
29
+ "cell_type": "markdown",
30
+ "metadata": {},
31
+ "source": [
32
+ "# Init ChromaDB and Embedding Function"
33
+ ]
34
+ },
35
+ {
36
+ "cell_type": "code",
37
+ "execution_count": null,
38
+ "metadata": {},
39
+ "outputs": [],
40
+ "source": [
41
+ "import chromadb\n",
42
+ "from chromadb.utils import embedding_functions\n",
43
+ "\n",
44
+ "from sentence_transformers import SentenceTransformer\n",
45
+ "from tqdm import tqdm\n",
46
+ "\n",
47
+ "\n",
48
+ "# Load a pre-trained sentence embedding model\n",
49
+ "model = SentenceTransformer(\"sentence-transformers/all-MiniLM-L6-v2\")\n",
50
+ "\n",
51
+ "def embedding_func(text):\n",
52
+ " return model.encode(text, convert_to_numpy=True) \n",
53
+ "\n",
54
+ "# Initialize ChromaDB (runs locally)\n",
55
+ "chroma_client = chromadb.PersistentClient(path=\"./chroma_db\") \n",
56
+ "\n",
57
+ "# Create or load a collection with metadata support\n",
58
+ "collection = chroma_client.get_or_create_collection(\n",
59
+ " name=\"oura_chunks\",\n",
60
+ " metadata={\"hnsw:space\": \"cosine\"} \n",
61
+ ")"
62
+ ]
63
+ },
64
+ {
65
+ "cell_type": "markdown",
66
+ "metadata": {},
67
+ "source": [
68
+ "# Add Chunks to ChromaDB"
69
+ ]
70
+ },
71
+ {
72
+ "cell_type": "code",
73
+ "execution_count": null,
74
+ "metadata": {},
75
+ "outputs": [],
76
+ "source": [
77
+ "# Insert chunks into ChromaDB\n",
78
+ "for chunk in tqdm(chunks):\n",
79
+ " unique_id = f\"{chunk['article_title'].replace(' ', '_')}_{chunk['chunk_id']}_{chunk['link']}\" \n",
80
+ "\n",
81
+ " # Check if the embedding ID already exists in the collection\n",
82
+ " existing_embedding = collection.get(ids=[unique_id])\n",
83
+ "\n",
84
+ " if existing_embedding[\"documents\"]:\n",
85
+ " print(f\"Embedding with ID {unique_id} already exists.\")\n",
86
+ " else:\n",
87
+ " collection.add(\n",
88
+ " ids=[unique_id], \n",
89
+ " embeddings=[embedding_func(chunk[\"chunk\"]).tolist()], \n",
90
+ " documents=[chunk[\"chunk\"]], \n",
91
+ " metadatas=[{\n",
92
+ " \"article_title\": chunk[\"article_title\"],\n",
93
+ " \"section_title\": chunk[\"section_title\"],\n",
94
+ " \"chunk_id\": chunk[\"chunk_id\"],\n",
95
+ " \"link\": chunk[\"link\"]\n",
96
+ " }]\n",
97
+ " )\n",
98
+ "\n",
99
+ "print(\"Chunks added successfully!\")\n"
100
+ ]
101
+ },
102
+ {
103
+ "cell_type": "code",
104
+ "execution_count": null,
105
+ "metadata": {},
106
+ "outputs": [],
107
+ "source": []
108
+ }
109
+ ],
110
+ "metadata": {
111
+ "kernelspec": {
112
+ "display_name": "ouraDemo",
113
+ "language": "python",
114
+ "name": "python3"
115
+ },
116
+ "language_info": {
117
+ "codemirror_mode": {
118
+ "name": "ipython",
119
+ "version": 3
120
+ },
121
+ "file_extension": ".py",
122
+ "mimetype": "text/x-python",
123
+ "name": "python",
124
+ "nbconvert_exporter": "python",
125
+ "pygments_lexer": "ipython3",
126
+ "version": "3.10.16"
127
+ }
128
+ },
129
+ "nbformat": 4,
130
+ "nbformat_minor": 2
131
+ }
extractOuraArticleDataChunks.ipynb ADDED
@@ -0,0 +1,249 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "metadata": {},
6
+ "source": [
7
+ "# Web Scrape Oura Ring App Content"
8
+ ]
9
+ },
10
+ {
11
+ "cell_type": "code",
12
+ "execution_count": null,
13
+ "metadata": {},
14
+ "outputs": [],
15
+ "source": [
16
+ "from selenium import webdriver\n",
17
+ "from selenium.webdriver.chrome.options import Options\n",
18
+ "from selenium.webdriver.common.by import By\n",
19
+ "import time\n",
20
+ "import re\n",
21
+ "\n",
22
+ "# Setup Chrome options\n",
23
+ "chrome_options = Options()\n",
24
+ "chrome_options.add_argument(\"--headless\")\n",
25
+ "chrome_options.add_argument(\"--disable-gpu\")\n",
26
+ "chrome_options.add_argument(\"--no-sandbox\")\n",
27
+ "chrome_options.add_argument(\"user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36\")\n",
28
+ "\n",
29
+ "# Initialize the WebDriver\n",
30
+ "driver = webdriver.Chrome(options=chrome_options)\n",
31
+ "\n",
32
+ "# Open the main page containing the list of articles\n",
33
+ "url = \"https://support.ouraring.com/hc/en-us/categories/27782541623059-Oura-App\"\n",
34
+ "driver.get(url)\n",
35
+ "time.sleep(3)\n",
36
+ "\n",
37
+ "# section_links = [link.get_attribute(\"href\") for link in driver.find_elements(By.XPATH, \"//h3[contains(@class, 'section-tree-title')]/a\")]\n",
38
+ "section_links = ['https://support.ouraring.com/hc/en-us/sections/9721296007827-App-Settings',\n",
39
+ " 'https://support.ouraring.com/hc/en-us/sections/9721508785171-App-Integrations',\n",
40
+ " 'https://support.ouraring.com/hc/en-us/sections/4415723882003-Readiness',\n",
41
+ " 'https://support.ouraring.com/hc/en-us/sections/4415717595155-Sleep',\n",
42
+ " 'https://support.ouraring.com/hc/en-us/sections/4415723875347-Activity',\n",
43
+ " 'https://support.ouraring.com/hc/en-us/sections/4415730373779-Insights-Reports',\n",
44
+ " 'https://support.ouraring.com/hc/en-us/sections/28332751996435-Heart-Health',\n",
45
+ " 'https://support.ouraring.com/hc/en-us/sections/17984010150803-Women-s-Health',\n",
46
+ " 'https://support.ouraring.com/hc/en-us/sections/4415723874451-Mindfulness-Meditation']\n",
47
+ "\n",
48
+ "articles_data = []\n",
49
+ "\n",
50
+ "# Function to split large sections into smaller RAG-friendly chunks\n",
51
+ "def chunk_text(text, chunk_size=512):\n",
52
+ " \n",
53
+ " # Split text into sentences and list items\n",
54
+ " list_item_pattern = re.compile(r\"^(\\d+\\.\\s+|- )\") \n",
55
+ " lines = text.split(\"\\n\") \n",
56
+ "\n",
57
+ " chunks = []\n",
58
+ " current_chunk = \"\"\n",
59
+ "\n",
60
+ " for line in lines:\n",
61
+ " line = line.strip()\n",
62
+ " if not line:\n",
63
+ " continue \n",
64
+ " \n",
65
+ " # Check if this line starts a new list item\n",
66
+ " is_list_item = bool(list_item_pattern.match(line))\n",
67
+ " \n",
68
+ " if is_list_item and current_chunk and len(current_chunk) + len(line) >= chunk_size:\n",
69
+ " chunks.append(current_chunk.strip())\n",
70
+ " current_chunk = line \n",
71
+ " elif len(current_chunk) + len(line) >= chunk_size:\n",
72
+ " chunks.append(current_chunk.strip())\n",
73
+ " current_chunk = line \n",
74
+ " else:\n",
75
+ " current_chunk += \"\\n\" + line \n",
76
+ " \n",
77
+ " if current_chunk:\n",
78
+ " chunks.append(current_chunk.strip()) \n",
79
+ "\n",
80
+ " return chunks\n",
81
+ "\n",
82
+ "\n",
83
+ "def extractSectionChunks(driver):\n",
84
+ " # Extract article title\n",
85
+ " title_element = driver.find_element(By.XPATH, \"//h1[contains(@class, 'article-title')]\")\n",
86
+ " article_title = title_element.text if title_element else \"No Title Found\"\n",
87
+ "\n",
88
+ " # Extract all content divs\n",
89
+ " content_elements = driver.find_elements(By.XPATH, \"//div[contains(@class, 'article-body')]/*\")\n",
90
+ "\n",
91
+ " chunks = []\n",
92
+ " intro_content = []\n",
93
+ " current_section = None\n",
94
+ " sections = []\n",
95
+ " chunk_id = 1 \n",
96
+ "\n",
97
+ " for element in content_elements:\n",
98
+ " tag_name = element.tag_name.lower()\n",
99
+ " text = element.text.strip()\n",
100
+ "\n",
101
+ " if not text:\n",
102
+ " continue \n",
103
+ "\n",
104
+ " # If a header is encountered, start a new section\n",
105
+ " if tag_name in [\"h2\"]:\n",
106
+ " # Store previous section\n",
107
+ " if current_section:\n",
108
+ " sections.append(current_section)\n",
109
+ " \n",
110
+ " # Start a new section\n",
111
+ " current_section = {\"section_title\": text, \"content\": []}\n",
112
+ " elif tag_name == \"ol\":\n",
113
+ " # Handle ordered (numbered) lists\n",
114
+ " list_items = element.find_elements(By.TAG_NAME, \"li\")\n",
115
+ " list_text = \"\\n\".join([f\"{idx+1}. {li.text.strip()}\" for idx, li in enumerate(list_items) if li.text.strip()])\n",
116
+ " if current_section:\n",
117
+ " current_section[\"content\"].append(list_text)\n",
118
+ " else:\n",
119
+ " intro_content.append(list_text)\n",
120
+ " elif tag_name == \"ul\":\n",
121
+ " # Handle unordered (bulleted) lists\n",
122
+ " list_items = element.find_elements(By.TAG_NAME, \"li\")\n",
123
+ " list_text = \"\\n\".join([f\"- {li.text.strip()}\" for li in list_items if li.text.strip()])\n",
124
+ " if current_section:\n",
125
+ " current_section[\"content\"].append(list_text)\n",
126
+ " else:\n",
127
+ " intro_content.append(list_text)\n",
128
+ " else:\n",
129
+ " if current_section:\n",
130
+ " current_section[\"content\"].append(text)\n",
131
+ " else:\n",
132
+ " # This is part of the introduction\n",
133
+ " intro_content.append(text)\n",
134
+ "\n",
135
+ " # Store last section\n",
136
+ " if current_section:\n",
137
+ " sections.append(current_section)\n",
138
+ "\n",
139
+ " # Store introduction **FIRST**\n",
140
+ " if intro_content:\n",
141
+ " intro_chunks = chunk_text(\"\\n\".join(intro_content))\n",
142
+ " for chunk in intro_chunks:\n",
143
+ " chunks.append({\n",
144
+ " \"article_title\": article_title,\n",
145
+ " \"section_title\": \"Introduction\",\n",
146
+ " \"chunk\": chunk,\n",
147
+ " \"chunk_id\": chunk_id, \n",
148
+ " \"link\": article_link\n",
149
+ " })\n",
150
+ " chunk_id += 1 \n",
151
+ "\n",
152
+ " # Store all sections\n",
153
+ " for section in sections:\n",
154
+ " section_chunks = chunk_text(\"\\n\".join(section[\"content\"]))\n",
155
+ " for chunk in section_chunks:\n",
156
+ " chunks.append({\n",
157
+ " \"article_title\": article_title,\n",
158
+ " \"section_title\": section[\"section_title\"],\n",
159
+ " \"chunk\": chunk,\n",
160
+ " \"chunk_id\": chunk_id, \n",
161
+ " \"link\": article_link\n",
162
+ " })\n",
163
+ " chunk_id += 1 \n",
164
+ " return article_title, chunks\n",
165
+ "\n",
166
+ "for section_link in section_links:\n",
167
+ " driver.get(section_link)\n",
168
+ " time.sleep(3)\n",
169
+ "\n",
170
+ " article_links = [link.get_attribute(\"href\") for link in driver.find_elements(By.XPATH, \"//li[contains(@class, 'article-list-item')]/a\")]\n",
171
+ "\n",
172
+ " for article_link in article_links:\n",
173
+ " driver.get(article_link)\n",
174
+ " time.sleep(3)\n",
175
+ "\n",
176
+ " try:\n",
177
+ " article_title,chunks = extractSectionChunks(driver)\n",
178
+ " articles_data.extend(chunks)\n",
179
+ " print(f\"Extracted: {article_title} - {len(chunks)} chunks\")\n",
180
+ " except Exception as e:\n",
181
+ " try:\n",
182
+ " driver.get(article_link)\n",
183
+ " sub_article_links = [link.get_attribute(\"href\") for link in driver.find_elements(By.XPATH, \"//li[contains(@class, 'article-list-item')]/a\")]\n",
184
+ " for sub_article_link in sub_article_links:\n",
185
+ " driver.get(sub_article_link)\n",
186
+ " time.sleep(3)\n",
187
+ " article_title,chunks = extractSectionChunks(driver)\n",
188
+ " articles_data.extend(chunks)\n",
189
+ " print(f\"Extracted: {article_title} - {len(chunks)} chunks\")\n",
190
+ " except Exception as e:\n",
191
+ " print(f\"Failed to extract content from {article_link}: {str(e)}\")\n",
192
+ "\n",
193
+ "\n",
194
+ "driver.quit()\n"
195
+ ]
196
+ },
197
+ {
198
+ "cell_type": "markdown",
199
+ "metadata": {},
200
+ "source": [
201
+ "# Save Chunks"
202
+ ]
203
+ },
204
+ {
205
+ "cell_type": "code",
206
+ "execution_count": null,
207
+ "metadata": {},
208
+ "outputs": [],
209
+ "source": [
210
+ "import json\n",
211
+ "# Save extracted articles to JSON\n",
212
+ "with open(\"oura_article_data_chunked_final.json\", \"w\", encoding=\"utf-8\") as f:\n",
213
+ " json.dump(articles_data, f, indent=4, ensure_ascii=False)\n",
214
+ "\n",
215
+ "driver.quit()\n",
216
+ "\n",
217
+ "print(\"Scraping complete. Data saved to 'oura_articles.json'\")"
218
+ ]
219
+ },
220
+ {
221
+ "cell_type": "code",
222
+ "execution_count": null,
223
+ "metadata": {},
224
+ "outputs": [],
225
+ "source": []
226
+ }
227
+ ],
228
+ "metadata": {
229
+ "kernelspec": {
230
+ "display_name": "ouraDemo",
231
+ "language": "python",
232
+ "name": "python3"
233
+ },
234
+ "language_info": {
235
+ "codemirror_mode": {
236
+ "name": "ipython",
237
+ "version": 3
238
+ },
239
+ "file_extension": ".py",
240
+ "mimetype": "text/x-python",
241
+ "name": "python",
242
+ "nbconvert_exporter": "python",
243
+ "pygments_lexer": "ipython3",
244
+ "version": "3.10.16"
245
+ }
246
+ },
247
+ "nbformat": 4,
248
+ "nbformat_minor": 2
249
+ }
extractOuraReddit.ipynb ADDED
@@ -0,0 +1,165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": null,
6
+ "metadata": {},
7
+ "outputs": [],
8
+ "source": [
9
+ "import praw\n",
10
+ "import os\n",
11
+ "from dotenv import load_dotenv\n",
12
+ "\n",
13
+ "load_dotenv()\n",
14
+ "\n",
15
+ "reddit = praw.Reddit(\n",
16
+ " client_id=os.getenv(\"REDDIT_CLIENT_ID\"),\n",
17
+ " client_secret=os.getenv(\"REDDIT_SECRET\"),\n",
18
+ " user_agent=\"OuraBot/1.0\"\n",
19
+ ")\n",
20
+ "\n",
21
+ "# Test connection\n",
22
+ "print(reddit.read_only) # Should print True if authentication is successful\n"
23
+ ]
24
+ },
25
+ {
26
+ "cell_type": "code",
27
+ "execution_count": null,
28
+ "metadata": {},
29
+ "outputs": [],
30
+ "source": [
31
+ "subreddit = reddit.subreddit(\"ouraring\")"
32
+ ]
33
+ },
34
+ {
35
+ "cell_type": "code",
36
+ "execution_count": null,
37
+ "metadata": {},
38
+ "outputs": [],
39
+ "source": [
40
+ "from textblob import TextBlob # For sentiment analysis\n",
41
+ "import time\n",
42
+ "\n",
43
+ "# Function to fetch posts with pagination\n",
44
+ "def fetch_posts_with_pagination(subreddit, query, limit=100, time_filter='all'):\n",
45
+ " posts = []\n",
46
+ " after = None\n",
47
+ " seen_posts = set() \n",
48
+ " while len(posts) < limit:\n",
49
+ " search_results = subreddit.search(query, limit=limit - len(posts), time_filter=time_filter, params={'after': after})\n",
50
+ " \n",
51
+ " for post in search_results:\n",
52
+ " if post.id not in seen_posts: \n",
53
+ " posts.append(post)\n",
54
+ " seen_posts.add(post.id) \n",
55
+ " if len(posts) >= limit:\n",
56
+ " break \n",
57
+ " \n",
58
+ " if not search_results:\n",
59
+ " break \n",
60
+ " \n",
61
+ " last_post = next(reversed(list(search_results)), None) \n",
62
+ " if last_post:\n",
63
+ " after = last_post.fullname \n",
64
+ " time.sleep(1) \n",
65
+ " \n",
66
+ " return posts\n",
67
+ "\n",
68
+ "\n",
69
+ "# Function to filter comments based on relevance, upvotes, length, and keywords\n",
70
+ "def filter_comments(comments, min_upvotes=5, keywords=None, min_length=50, max_length=500):\n",
71
+ " filtered_comments = []\n",
72
+ " for comment in comments:\n",
73
+ " if comment.score < min_upvotes:\n",
74
+ " continue\n",
75
+ " \n",
76
+ " if keywords and not any(keyword.lower() in comment.body.lower() for keyword in keywords):\n",
77
+ " continue\n",
78
+ " \n",
79
+ " comment_length = len(comment.body.split())\n",
80
+ " if comment_length < min_length or comment_length > max_length:\n",
81
+ " continue\n",
82
+ " \n",
83
+ " sentiment = TextBlob(comment.body).sentiment.polarity\n",
84
+ " if sentiment < 0: # Only keep positive or neutral comments\n",
85
+ " continue\n",
86
+ " \n",
87
+ " filtered_comments.append(comment)\n",
88
+ " \n",
89
+ " filtered_comments = sorted(filtered_comments, key=lambda x: x.score, reverse=True)\n",
90
+ " return filtered_comments\n",
91
+ "\n",
92
+ "\n",
93
+ "# Define relevant keywords for Oura app support (specific to the app and troubleshooting)\n",
94
+ "keywords = [\n",
95
+ " 'fix', 'solution', 'troubleshoot', 'support', 'sync', 'bug', 'pairing', 'error', 'customer service', \n",
96
+ " 'installation', 'crash', 'not working', 'app issue', 'reset', 'login issue', 'help'\n",
97
+ "]\n",
98
+ "\n",
99
+ "\n",
100
+ "# Function to determine if a post is high-quality and relevant to Oura app\n",
101
+ "def is_acceptable_post(post):\n",
102
+ " if post.score <= 0 or post.upvote_ratio < 0.6 or post.over_18:\n",
103
+ " return False\n",
104
+ " \n",
105
+ " if post.link_flair_text and \"Support\" not in post.link_flair_text:\n",
106
+ " return False\n",
107
+ " \n",
108
+ " if post.num_comments < 3:\n",
109
+ " return False\n",
110
+ " \n",
111
+ " from datetime import datetime\n",
112
+ " if (datetime.utcnow() - datetime.utcfromtimestamp(post.created_utc)).days > 30:\n",
113
+ " return False\n",
114
+ " \n",
115
+ " return True\n",
116
+ "\n",
117
+ "# Fetch posts with pagination\n",
118
+ "subreddit = reddit.subreddit(\"ouraring\")\n",
119
+ "posts = fetch_posts_with_pagination(subreddit, query, limit=1000, time_filter='all')\n",
120
+ "\n",
121
+ "# Process and display the posts and their filtered comments\n",
122
+ "for post in posts:\n",
123
+ " if is_acceptable_post(post): \n",
124
+ " print(f\"Post Title: {post.title}\")\n",
125
+ " print(f\"Post Link: {post.url}\")\n",
126
+ " print(f\"Upvotes: {post.score}\")\n",
127
+ "\n",
128
+ " post.comments.replace_more(limit=0)\n",
129
+ " filtered_comments = filter_comments(post.comments, min_upvotes=5, keywords=keywords, min_length=50, max_length=500)\n",
130
+ "\n",
131
+ " for comment in filtered_comments:\n",
132
+ " print(f\" Comment Score: {comment.score}\")\n",
133
+ " print(f\" Comment: {comment.body}\\n\")\n"
134
+ ]
135
+ },
136
+ {
137
+ "cell_type": "code",
138
+ "execution_count": null,
139
+ "metadata": {},
140
+ "outputs": [],
141
+ "source": []
142
+ }
143
+ ],
144
+ "metadata": {
145
+ "kernelspec": {
146
+ "display_name": "ouraDemo",
147
+ "language": "python",
148
+ "name": "python3"
149
+ },
150
+ "language_info": {
151
+ "codemirror_mode": {
152
+ "name": "ipython",
153
+ "version": 3
154
+ },
155
+ "file_extension": ".py",
156
+ "mimetype": "text/x-python",
157
+ "name": "python",
158
+ "nbconvert_exporter": "python",
159
+ "pygments_lexer": "ipython3",
160
+ "version": "3.10.16"
161
+ }
162
+ },
163
+ "nbformat": 4,
164
+ "nbformat_minor": 2
165
+ }
oura_article_data_chunked_final.json ADDED
The diff for this file is too large to render. See raw diff
 
runOuraAppAssistant.ipynb ADDED
@@ -0,0 +1,248 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "metadata": {},
6
+ "source": [
7
+ "# Run the Oura App Assistant"
8
+ ]
9
+ },
10
+ {
11
+ "cell_type": "markdown",
12
+ "metadata": {},
13
+ "source": [
14
+ "Here, we retrieve the most relevant chunks based on the user's query and incorporate into a prompt sent to the LLM for a response."
15
+ ]
16
+ },
17
+ {
18
+ "cell_type": "code",
19
+ "execution_count": null,
20
+ "metadata": {},
21
+ "outputs": [],
22
+ "source": [
23
+ "import ollama\n",
24
+ "from openai import OpenAI\n",
25
+ "import chromadb\n",
26
+ "from sentence_transformers import SentenceTransformer\n",
27
+ "import os\n",
28
+ "from dotenv import load_dotenv\n",
29
+ "\n",
30
+ "load_dotenv()\n",
31
+ "\n",
32
+ "# Use a basic sentence transformer finetuned for text similarity tasks\n",
33
+ "model = SentenceTransformer(\"sentence-transformers/all-MiniLM-L6-v2\")\n",
34
+ "\n",
35
+ "ollama.pull(\"mistral\")\n",
36
+ "\n",
37
+ "def embedding_func(text):\n",
38
+ " return model.encode(text, convert_to_numpy=True) \n",
39
+ "\n",
40
+ "# Initialize ChromaDB (runs locally)\n",
41
+ "chroma_client = chromadb.PersistentClient(path=\"./chroma_db\") \n",
42
+ "\n",
43
+ "# Create or load a collection with metadata support\n",
44
+ "collection = chroma_client.get_or_create_collection(\n",
45
+ " name=\"oura_chunks\",\n",
46
+ " metadata={\"hnsw:space\": \"cosine\"} \n",
47
+ ")\n",
48
+ "\n",
49
+ "# Initialize LLM (Ollama Mistral model)\n",
50
+ "def query_ollama(query, context):\n",
51
+ " # Construct prompt with instructions, context, and query\n",
52
+ " prompt = (\n",
53
+ " \"You are an expert assistant answering questions about the Oura Ring app.\\n\"\n",
54
+ " \"Based on the following context, please answer the user's question.\\n\"\n",
55
+ " \"Only use the provided context for your response. If you cannot find a relevant answer, state that the query is out of scope.\\n\"\n",
56
+ " \"Cite the most relevant article, section, and link that are directly found in the provided context.\\n\"\n",
57
+ " \"Do not provide links that are not found in the provided context.\\n\\n\"\n",
58
+ " f\"{context}\\n\\n\"\n",
59
+ " f\"User's Question: {query}\"\n",
60
+ " )\n",
61
+ " # Submit prompt and get response\n",
62
+ " response = ollama.chat(model=\"mistral\", messages=[{\"role\": \"user\", \"content\": prompt}])\n",
63
+ " return response[\"message\"][\"content\"]\n",
64
+ "\n",
65
+ "def query_openai(query, context):\n",
66
+ " client = OpenAI(api_key=os.getenv(\"OPENAI_API_KEY\"))\n",
67
+ "\n",
68
+ " # Add instructions to the system message\n",
69
+ " system_message = {\"role\": \"system\", \n",
70
+ " \"content\": \"You are an expert assistant answering questions about the Oura Ring app.\\n\"\n",
71
+ " \"Use only the provided context for your responses.\\n\"\n",
72
+ " \"If you do not find relevant information, clearly state that the question is out of scope.\\n\"\n",
73
+ " \"Always cite the most relevant article, section, and link that are directly found in the provided context.\\n\"\n",
74
+ " \"Do not provide links that are not found in the provided context.\"}\n",
75
+ " \n",
76
+ " # Add the user's input\n",
77
+ " user_message = {\"role\": \"user\", \"content\": query}\n",
78
+ "\n",
79
+ " # Submit system message for response\n",
80
+ " completion = client.chat.completions.create(\n",
81
+ " model=\"gpt-4\", \n",
82
+ " messages=[\n",
83
+ " system_message, \n",
84
+ " {\"role\": \"assistant\", \"content\": context}, \n",
85
+ " user_message \n",
86
+ " ],\n",
87
+ " )\n",
88
+ " \n",
89
+ " # Output the chatbot's response\n",
90
+ " return completion.choices[0].message.content\n",
91
+ "\n",
92
+ "# Query ChromaDB for relevant chunks\n",
93
+ "def retrieve_relevant_chunks(query, n_results=5):\n",
94
+ " query_embedding = embedding_func(query).tolist() \n",
95
+ " results = collection.query(\n",
96
+ " query_embeddings=[query_embedding], \n",
97
+ " n_results=n_results\n",
98
+ " )\n",
99
+ " return results\n",
100
+ "\n",
101
+ "# Perform RAG (Retrieve + Generate) with Ollama/OpenAI\n",
102
+ "def perform_rag(query, provider=\"ollama\"):\n",
103
+ " # Retrieve relevant chunks from ChromaDB\n",
104
+ " results = retrieve_relevant_chunks(query, n_results=5)\n",
105
+ " \n",
106
+ " # Combine the retrieved chunks for LLM input\n",
107
+ " relevant_chunks = []\n",
108
+ " for document, metadata, distance in zip(results[\"documents\"][0],results[\"metadatas\"][0],results[\"distances\"][0]):\n",
109
+ " # Apply relatively lenient filter to return most relevant chunks\n",
110
+ " if distance > 0.75:\n",
111
+ " continue\n",
112
+ " # Structure chunks to include all relevant metadata\n",
113
+ " relevant_chunks.append(f'\\nArticle Title: {metadata[\"article_title\"]}\\nSection Title: {metadata[\"section_title\"]}\\nArticle Link: {metadata[\"link\"]}\\nRelevance Score: {100*round(1 - distance, 2)}\\nArticle Content: {document}\\n')\n",
114
+ " if len(relevant_chunks) < 1:\n",
115
+ " context = f'\\nArticle Title: No Relevant Article\\nSection Title: No Relevant Article\\nArticle Link: No Relevant Article\\nRelevance Score: No Relevant Article\\nArticle Content: No Relevant Article\\n'\n",
116
+ " else:\n",
117
+ " context = \"\\n\".join(relevant_chunks) \n",
118
+ "\n",
119
+ " if provider == \"ollama\":\n",
120
+ " # Query Ollama Mistral for a response\n",
121
+ " answer = query_ollama(query, context)\n",
122
+ " elif provider == \"openai\":\n",
123
+ " # Query OpenAI gpt-4 for a response\n",
124
+ " answer = query_openai(query, context)\n",
125
+ " else:\n",
126
+ " print(\"Invalid ChatBot provider, please select either ollama or openai!\")\n",
127
+ " \n",
128
+ " return answer\n",
129
+ "\n"
130
+ ]
131
+ },
132
+ {
133
+ "cell_type": "markdown",
134
+ "metadata": {},
135
+ "source": [
136
+ "# Qualitative Evaluation"
137
+ ]
138
+ },
139
+ {
140
+ "cell_type": "markdown",
141
+ "metadata": {},
142
+ "source": [
143
+ "For the qualitative evaluation, I provide each model with three sets of queries: 1) Relevant Queries, 2) Non-Relevant Queries, 3) Troubleshooting Queries, and 4) Subjective Queries. These queries highlight the model's ability to correctly answer questions relevant to the Oura app, identify when the question is out of scope, provide some coaching for troubleshooting and summarize subjective queries about the Oura App. "
144
+ ]
145
+ },
146
+ {
147
+ "cell_type": "code",
148
+ "execution_count": null,
149
+ "metadata": {},
150
+ "outputs": [],
151
+ "source": [
152
+ "\n",
153
+ "providers = {\"Ollama (Mistral) Evaluation\": \"ollama\", \n",
154
+ " \"Open AI (gpt-4) Evaluation\": \"openai\"}\n",
155
+ "query_types = {\"Relevant Queries\": [\n",
156
+ "\"How can I set the language for the Oura App?\",\n",
157
+ "\"Does the Oura App provide information about sleep quality?\"],\n",
158
+ "\"Non-Relevant Queries\": [\"How can I order a hamburger?\",\n",
159
+ " \"How can I order an Oura ring for my dog?\"],\n",
160
+ "\"Troubleshooting Queries\": [\"What should I do if my Oura Ring is not syncing with the Oura App?\",\n",
161
+ " \"What should I do if my Oura Ring is not syncing with Apple Health?\"],\n",
162
+ "\"Subjective Queries\": [\"Do users generally prefer the Oura App instead of the Apple Health App?\"]}"
163
+ ]
164
+ },
165
+ {
166
+ "cell_type": "code",
167
+ "execution_count": null,
168
+ "metadata": {},
169
+ "outputs": [],
170
+ "source": [
171
+ "\n",
172
+ "for provider in providers.keys():\n",
173
+ " print(\"================================================================================\")\n",
174
+ " print(provider)\n",
175
+ " for query_type in query_types.keys():\n",
176
+ " print(f\"********************** {query_type} **********************\")\n",
177
+ " for idx,question in enumerate(query_types[query_type]):\n",
178
+ " print(\"------------------------------------------------------------\")\n",
179
+ " answer = perform_rag(question, providers[provider])\n",
180
+ " output_string = f\"{idx+1}) Question: {question}\\nAnswer: {answer}\\n\\n\"\n",
181
+ " print(output_string)\n",
182
+ "\n"
183
+ ]
184
+ },
185
+ {
186
+ "cell_type": "markdown",
187
+ "metadata": {},
188
+ "source": [
189
+ "# Local Deployment to Gradio"
190
+ ]
191
+ },
192
+ {
193
+ "cell_type": "code",
194
+ "execution_count": null,
195
+ "metadata": {},
196
+ "outputs": [],
197
+ "source": [
198
+ "import gradio as gr\n",
199
+ "\n",
200
+ "def ask_oura_assistant(user_input, provider):\n",
201
+ " if \"ollama\" in provider.lower():\n",
202
+ " return perform_rag(user_input, \"ollama\")\n",
203
+ " elif \"openai\" in provider.lower():\n",
204
+ " return perform_rag(user_input, \"openai\")\n",
205
+ "\n",
206
+ "demo = gr.Interface(\n",
207
+ " fn=ask_oura_assistant, \n",
208
+ " inputs=[\n",
209
+ " gr.Textbox(label=\"Ask a question...\"), \n",
210
+ " gr.Dropdown([\"OpenAI (gpt-4)\", \"Ollama (Mistral)\"], label=\"Select Provider\") \n",
211
+ " ],\n",
212
+ " outputs=\"text\", \n",
213
+ " title=\"Oura Ring App Assistant\",\n",
214
+ " description=\"Ask me anything about the Oura Ring app! Select your provider before submitting.\")\n",
215
+ "\n",
216
+ "demo.launch()"
217
+ ]
218
+ },
219
+ {
220
+ "cell_type": "code",
221
+ "execution_count": null,
222
+ "metadata": {},
223
+ "outputs": [],
224
+ "source": []
225
+ }
226
+ ],
227
+ "metadata": {
228
+ "kernelspec": {
229
+ "display_name": "ouraDemo",
230
+ "language": "python",
231
+ "name": "python3"
232
+ },
233
+ "language_info": {
234
+ "codemirror_mode": {
235
+ "name": "ipython",
236
+ "version": 3
237
+ },
238
+ "file_extension": ".py",
239
+ "mimetype": "text/x-python",
240
+ "name": "python",
241
+ "nbconvert_exporter": "python",
242
+ "pygments_lexer": "ipython3",
243
+ "version": "3.10.16"
244
+ }
245
+ },
246
+ "nbformat": 4,
247
+ "nbformat_minor": 2
248
+ }