-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathprompts.py
More file actions
527 lines (409 loc) · 15.4 KB
/
Copy pathprompts.py
File metadata and controls
527 lines (409 loc) · 15.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
import json
import re
from abc import ABC, abstractmethod
from contextlib import nullcontext
from typing import Any, Dict, List, Optional, TypedDict
import openai
import pandas as pd
import streamlit as st
import tiktoken
class BaseTask(ABC):
@property
@abstractmethod
def name(self) -> str:
"""Returns the name of the task (library/framework)."""
return NotImplemented
@abstractmethod
def get_task_prompt(
self,
task_name: str,
description: str,
) -> str:
return NotImplemented
@abstractmethod
def handle_output(self, task_output: Any) -> Optional[pd.DataFrame]:
pass
_PLOTLY_PROMPT = """
import pandas as pd
import numpy as np
from typing import *
import plotly
def {method_name}(data: pd.DataFrame) -> Union["plotly.graph_objs.Figure", "plotly.graph_objs.Data"]:
\"\"\"
{description}
Visualize the data using the Plotly library.
Args:
data (pd.DataFrame): The data to visualize.
Returns:
Union[plotly.graph_objs.Figure, plotly.graph_objs.Data]: The plotly visualization figure object.
\"\"\"
# All task-specific imports here:
import plotly
{{Task implementation}}
"""
class PlotlyVisualizationTask(BaseTask):
"""A task that visualizes data using Plotly."""
@property
def name(self) -> str:
return "Plotly"
def get_task_prompt(
self,
task_name: str,
description: str,
) -> str:
return _PLOTLY_PROMPT.format(
method_name=task_name,
description=description,
)
def handle_output(self, task_output: Any) -> None:
import streamlit as st
st.plotly_chart(task_output, use_container_width=True)
_MATPLOTLIB_PROMPT = """
import pandas as pd
import numpy as np
from typing import *
from matplotlib.figure import Figure
def {method_name}(data: pd.DataFrame) -> "Figure":
\"\"\"
{description}
Visualize the data using the Matplotlib library.
Args:
data (pd.DataFrame): The data to visualize.
Returns:
matplotlib.figure.Figure: The matplotlib visualization figure.
\"\"\"
# All task-specific imports here:
import matplotlib
{{Task implementation}}
"""
class MatplotlibVisualizationTask(BaseTask):
"""A task that visualizes data using Matplotlib."""
@property
def name(self) -> str:
return "Matplotlib"
def get_task_prompt(
self,
task_name: str,
description: str,
) -> str:
return _MATPLOTLIB_PROMPT.format(
method_name=task_name,
description=description,
)
def handle_output(self, task_output: Any) -> None:
import streamlit as st
st.pyplot(task_output)
_ALTAIR_PROMPT = """
import pandas as pd
import numpy as np
from typing import *
import altair as alt
def {method_name}(data: pd.DataFrame) -> "alt.Chart":
\"\"\"
{description}
Visualize the data using the Altair library.
Args:
data (pd.DataFrame): The data to visualize.
Returns:
alt.Chart: The altair visualization chart.
\"\"\"
# All task-specific imports here:
import altair as alt
{{Task implementation}}
"""
class AltairVisualizationTask(BaseTask):
"""A task that visualizes data using Altair."""
@property
def name(self) -> str:
return "Altair"
def get_task_prompt(
self,
task_name: str,
description: str,
) -> str:
return _ALTAIR_PROMPT.format(
method_name=task_name,
description=description,
)
def handle_output(self, task_output: Any) -> None:
import streamlit as st
st.altair_chart(task_output, use_container_width=True)
TOOLS: List[BaseTask] = [
PlotlyVisualizationTask(),
AltairVisualizationTask(),
MatplotlibVisualizationTask(),
]
def extract_first_code_block(markdown_text: str) -> str:
if (
"```" not in markdown_text
and "return" in markdown_text
and "def" in markdown_text
):
# Assume that everything is code
return markdown_text
pattern = r"(?<=```).+?(?=```)"
# Use re.DOTALL flag to make '.' match any character, including newlines
first_code_block = re.search(pattern, markdown_text, flags=re.DOTALL)
code = first_code_block.group(0) if first_code_block else ""
code = code.strip().lstrip("python").strip()
return code
def num_tokens_from_messages(
messages: List[dict], model: str = "gpt-3.5-turbo-0301"
) -> int:
"""Returns the number of tokens used by a list of messages."""
try:
encoding = tiktoken.encoding_for_model(model)
except KeyError:
encoding = tiktoken.get_encoding("cl100k_base")
if model == "gpt-3.5-turbo-0301": # note: future models may deviate from this
num_tokens = 0
for message in messages:
num_tokens += (
4 # every message follows <im_start>{role/name}\n{content}<im_end>\n
)
for key, value in message.items():
num_tokens += len(encoding.encode(value))
if key == "name": # if there's a name, the role is omitted
num_tokens += -1 # role is always required and always 1 token
num_tokens += 2 # every reply is primed with <im_start>assistant
return num_tokens
else:
raise NotImplementedError(
f"""num_tokens_from_messages() is not presently implemented for model {model}.
See https://github.com/openai/openai-python/blob/main/chatml.md for information on how messages are converted to tokens."""
)
def get_df_types_info(df: pd.DataFrame = None) -> str:
column_info = {
"dtype": [str(dtype) for dtype in df.dtypes],
"inferred dtype": [
pd.api.types.infer_dtype(column) for _, column in df.items()
],
"missing values": df.isna().sum().to_list(),
}
return pd.DataFrame(
column_info,
index=df.columns,
).to_csv()
def get_column_descriptions_info(column_desc_df: pd.DataFrame) -> str:
return "Column descriptions: \n\n" + column_desc_df.to_csv()
def get_df_stats_info(df: pd.DataFrame) -> str:
return df.describe().to_csv()
def get_df_sample_info(df: pd.DataFrame, sample_size: int = 20) -> str:
return df.sample(min(len(df), sample_size), random_state=1).to_csv()
def get_dataset_description_prompt(
dataset_df: pd.DataFrame,
column_desc_df: Optional[pd.DataFrame] = None,
light: bool = False,
) -> str:
prompt = f"""
I have a dataset (as Pandas DataFrame) that contains data with the following characteristics:
Column data types:
{get_df_types_info(dataset_df)}
{get_column_descriptions_info(column_desc_df) if column_desc_df is not None else ""}
Data sample:
{get_df_sample_info(dataset_df, sample_size=10 if light else 20)}
"""
if light:
return prompt
return (
prompt
+ f"""
Data statistics:
{get_df_stats_info(dataset_df)}
"""
)
class DataDescription(TypedDict):
data_description: str
columns: List[Dict[str, str]]
observations: List[str]
chart_ideas: List[str]
@st.cache_data(show_spinner=False)
def get_data_description(
dataset_df: pd.DataFrame,
openai_model: str = "gpt-3.5-turbo",
show_system_messages: bool = False,
) -> DataDescription:
user_prompt = f"""
{get_dataset_description_prompt(dataset_df)}
Please provide the following information based on this JSON template:
{{
"data_description": "{{A concise description of the dataset}}",
"columns": [
{{ "name": "{{column name}}", "description": "{{A short description about the column}}" }},
]
"observations": [
"{{Interesting or surprising observations about the data}}",
]
"chart_ideas": [
"{{Concise description of your best chart or visualization idea}}",
"{{Concise description of your second best chart or visualization idea}}",
"{{Concise description of the third best chart or visualization idea}}",
"{{Concise description of the fourth best chart or visualization idea}}",
]
}}
Please only respond with a valid JSON and fill out all {{placeholders}}:
"""
messages = [
{
"role": "system",
"content": "You are a helpful assistant that helps with describing data. The user will share some information about a Pandas dataframe, and you will create a description of the data based on a user-provided JSON template. Please provide your answer as valid JSON.",
},
{"role": "user", "content": user_prompt},
]
print("Prompt tokens", num_tokens_from_messages(messages))
completion = openai.ChatCompletion.create(model=openai_model, messages=messages)
response_json = json.loads(completion.choices[0].message.content)
if show_system_messages:
with st.expander("System messages"):
for message in messages:
st.markdown(message["content"])
st.divider()
st.markdown("**GPT Response:**")
st.json(response_json)
st.divider()
st.markdown(f"**User tokens:** {num_tokens_from_messages(messages)}")
return response_json
class TaskSummary(TypedDict):
name: str
method_name: str
task_description: str
emoji: str
@st.cache_data(show_spinner=False)
def get_task_summary(
dataset_df: pd.DataFrame,
task_instruction: str,
openai_model: str = "gpt-3.5-turbo",
column_desc_df: Optional[pd.DataFrame] = None,
show_system_messages: bool = False,
) -> TaskSummary:
user_prompt = f"""
{get_dataset_description_prompt(dataset_df, column_desc_df)}
Based on this data, I have the following task instruction:
{task_instruction}
Please provide me the following information based on the task instruction:
{{
"name": "{{Short Task Name}}",
"method_name": "{{python_method_name}}",
"task_description": "{{a concise description of the task}}",
"emoji": "{{a descriptive emoji}}",
}}
The JSON needs to be compatible with the following TypeDict:
```python
class TaskSummary(TypedDict):
name: str
method_name: str
task_description: str
emoji: str
```
Please only respond with a valid JSON:
"""
messages = [
{
"role": "system",
"content": "You are a helpful assistant. The user will share some characteristics of a dataset and a data exploration or visualization task instructions that the user likes to perform on this dataset. You will create some descriptive information about the task instruction based on a provided JSON template. Please only respond with a valid JSON.",
},
{"role": "user", "content": user_prompt},
]
print("Prompt tokens", num_tokens_from_messages(messages))
completion = openai.ChatCompletion.create(model=openai_model, messages=messages)
response_json = json.loads(completion.choices[0].message.content)
if show_system_messages:
with st.expander("System messages"):
for message in messages:
st.markdown(message["content"])
st.divider()
st.markdown("**GPT Response:**")
st.json(response_json)
st.divider()
st.markdown(f"**User tokens:** {num_tokens_from_messages(messages)}")
return response_json
@st.cache_data(show_spinner=False)
def get_task_code(
dataset_df: pd.DataFrame,
task_instruction: str,
task_code_template: str,
task_library: str,
openai_model: str = "gpt-3.5-turbo",
column_desc_df: Optional[pd.DataFrame] = None,
show_system_messages: bool = False,
) -> str:
user_prompt = f"""
{get_dataset_description_prompt(dataset_df, column_desc_df)}
Based on this data, please create a function that performs the following Data Visualization task with {task_library}:
{task_instruction}
Please use the following template and implement all {{placeholders}}:
```
{task_code_template}
```
Implement all {{placeholders}}, make sure that the Python code is valid. Please put the code in a markdown code block (```) and ONLY respond with the code:
"""
messages = [
{
"role": "system",
"content": "You are a helpful assistant that helps create python functions to perform data exploration and visualization tasks on a Pandas dataframe. The user will provide a description about the dataframe as well as a code template and instructions. Please only answer with valid Python code.",
# "content": "You are a helpful assistant that helps create python functions to perform data visualizations on a Pandas dataframe. The user will provide a description about the dataframe as well as a code template and instructions. Please only answer with valid Python code.",
},
{"role": "user", "content": user_prompt},
]
print("Prompt tokens", num_tokens_from_messages(messages))
completion = openai.ChatCompletion.create(model=openai_model, messages=messages)
response = completion.choices[0].message.content.strip()
if show_system_messages:
with st.expander("System messages"):
for message in messages:
st.markdown(message["content"])
st.divider()
st.markdown("**GPT Response:**")
st.markdown(response)
st.divider()
st.markdown(f"**User tokens:** {num_tokens_from_messages(messages)}")
code_block = extract_first_code_block(response)
if not code_block and response:
st.info(response, icon="🤖")
return code_block
@st.cache_data(show_spinner=False)
def get_modified_code(
task_code: str,
instruction: str,
dataset_df: pd.DataFrame,
exception_message: Optional[str] = None,
column_desc_df: Optional[pd.DataFrame] = None,
openai_model: str = "gpt-3.5-turbo",
show_system_messages: bool = False,
) -> str:
user_prompt = f"""
{get_dataset_description_prompt(dataset_df, column_desc_df, light=True)}
Based on this data, I implemented the following python function:
```python
{task_code}
```
{"But it throws the following exception: " + exception_message if exception_message else ""}
Please modify this code based on the following instruction:
```
{instruction}
```
It is very important to use the same function name and signature from the original code. Put the code in a markdown code block (```) and only respond with entire code that includes the modifications:
"""
messages = [
{
"role": "system",
"content": "You are a helpful assistant that helps to modify python functions that perform data exploration or visualization tasks on a Pandas dataframe. The user will provide a function implementation and a modification instruction. Please modify the code based on the instruction and only answer with valid Python code.",
},
{"role": "user", "content": user_prompt},
]
print("Prompt tokens", num_tokens_from_messages(messages))
completion = openai.ChatCompletion.create(model=openai_model, messages=messages)
response = completion.choices[0].message.content.strip()
if show_system_messages:
with st.expander("System messages"):
for message in messages:
st.markdown(message["content"])
st.divider()
st.markdown("**GPT Response:**")
st.markdown(response)
st.divider()
st.markdown(f"**User tokens:** {num_tokens_from_messages(messages)}")
code_block = extract_first_code_block(response)
if not code_block and response:
st.info(response, icon="🤖")
return code_block