diff --git a/Pipfile b/Pipfile
index 6069200..0f8e7ef 100644
--- a/Pipfile
+++ b/Pipfile
@@ -22,7 +22,7 @@ pytest = "==5.3.4"
pyre-check = "==0.0.41"
## like the Unix `make` but better
invoke = "==1.4.1"
-
+memory-profiler = "*"
[packages]
# REST API
diff --git a/Pipfile.lock b/Pipfile.lock
index 6575e89..e55e8ff 100644
--- a/Pipfile.lock
+++ b/Pipfile.lock
@@ -1,7 +1,7 @@
{
"_meta": {
"hash": {
- "sha256": "fb30d39142d3cc83d8909d9f4f4648a60ac33d4ec3a5a94d8dac7b90ef727a24"
+ "sha256": "c1f663e58339a2e67ba7d26a44722711969699a9998316437dcaef26cbbe8b80"
},
"pipfile-spec": 6,
"requires": {
@@ -826,6 +826,13 @@
],
"version": "==0.6.1"
},
+ "memory-profiler": {
+ "hashes": [
+ "sha256:23b196f91ea9ac9996e30bfab1e82fecc30a4a1d24870e81d1e81625f786a2c3"
+ ],
+ "index": "pypi",
+ "version": "==0.57.0"
+ },
"more-itertools": {
"hashes": [
"sha256:5dd8bcf33e5f9513ffa06d5ad33d78f31e1931ac9a18f33d37e77a180d393a7c",
diff --git a/QA.py b/QA.py
index 250b065..06ecf0b 100644
--- a/QA.py
+++ b/QA.py
@@ -9,6 +9,8 @@ from Entity.Sections import Sections
from database_wrapper import NimbusMySQLAlchemy
from pandas import read_csv
+from memory_profiler import profile
+
Extracted_Vars = Dict[str, Any]
DB_Data = Dict[str, Any]
DB_Query = Callable[[Extracted_Vars], DB_Data]
@@ -33,6 +35,7 @@ class QA:
A class for wrapping functions used to answer a question.
"""
+ @profile
def __init__(self, q_format, db_query, format_answer):
"""
Args:
@@ -55,6 +58,7 @@ class QA:
def _format_answer(self, extracted_vars, db_data):
return self.format_answer(extracted_vars, db_data)
+ @profile
def answer(self, extracted_vars):
db_data = self._get_data_from_db(extracted_vars)
return self._format_answer(extracted_vars, db_data)
@@ -66,6 +70,7 @@ class QA:
return hash(self.q_format)
+@profile
def create_qa_mapping(qa_list):
"""
Creates a dictionary whose values are QA objects and keys are the question
@@ -186,6 +191,7 @@ def yes_no(a_format, pred=None):
return functools.partial(_yes_no, a_format, pred)
+@profile
def generate_fact_QA(csv):
df = read_csv(csv)
text_in_brackets = r'\[[^\[\]]*\]'
diff --git a/flask_api.py b/flask_api.py
index d6478c5..34fe17b 100755
--- a/flask_api.py
+++ b/flask_api.py
@@ -18,6 +18,8 @@ from modules.validators import WakeWordValidator, WakeWordValidatorError
from nimbus import Nimbus
+from memory_profiler import profile
+
BAD_REQUEST = 400
SUCCESS = 200
@@ -44,6 +46,7 @@ def generate_session_token() -> str:
return "SOME_NEW_TOKEN"
+@profile
@app.route('/ask', methods=['POST'])
def handle_question():
"""
diff --git a/nimbus.py b/nimbus.py
index 7d6bdfc..729cf62 100644
--- a/nimbus.py
+++ b/nimbus.py
@@ -9,14 +9,17 @@ from werkzeug.exceptions import BadRequestKeyError
from QA import create_qa_mapping, generate_fact_QA
from nimbus_nlp.NIMBUS_NLP import NIMBUS_NLP
+from memory_profiler import profile
class Nimbus:
+ @profile
def __init__(self):
self.qa_dict = create_qa_mapping(
generate_fact_QA("q_a_pairs.csv")
)
+ @profile
def answer_question(self, question):
ans_dict = NIMBUS_NLP.predict_question(question)
print(ans_dict)
(END)
Bug Description
The Heroku Standard 1x Dyno has a 512 MB memory quota and we have hit the limit after just
/askendpointScreenshots
screenshots
Proposed solutions
1. switch to GCP and choose an appropriately sized virtual machine
.github/workflows/*.ymlto handle GCP deployment2. scale the heroku dyno to standard-2x or something else
This would get expensive, quickly...
3. identify memory usage & performance improvements for our code
maybe avoid the use of
pandasinQAand useimport csvinsteadonly
read_csvis importedapi/QA.py
Line 10 in 48336b0
reference @zpdeng 's usage of
csvmoduleapi/database_wrapper.py
Lines 426 to 433 in 48336b0
it may be a good idea to actually test if
pandasis truly hogging memorymaybe
spacy?maybe
nltk?maybe
SQLAlchemy?maybe
Flask?maybe
gcloud?maybe
another_package_expected_to_be_large?consider this output of
python3 -m memory_profiler flask_api.pymemory profile
git diff