Spaces:

magnolia-psychometrics
/

synth-net

Running

App Files Files Community

bjorn-hommel commited on May 15

Commit

cf1362c

1 Parent(s): e49e13a

added psyctest references

Browse files

Files changed (7) hide show

.gitignore +3 -1
all_mpnet_base_v2.enc +2 -2
app.py +132 -73
modeling.py +86 -45
psisent.enc +2 -2
requirements.txt +2 -0
surveybot3000.enc +2 -2

.gitignore CHANGED Viewed

@@ -1,3 +1,4 @@
 .env
 db.parquet
 preprocess.py
@@ -5,4 +6,5 @@ encrypt.py
 surveybot3000.parquet
 psisent.parquet
 all_mpnet_base_v2.parquet
-__pycache__

+psyctest_doi.parquet
 .env
 db.parquet
 preprocess.py
 surveybot3000.parquet
 psisent.parquet
 all_mpnet_base_v2.parquet
+__pycache__
+**tmp.**

all_mpnet_base_v2.enc CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:1835bad0cd24cc9019803c0589430fa4af320a16da456c364c22f21d51590ed2
-size 97110456

 version https://git-lfs.github.com/spec/v1
+oid sha256:87cc6e38ec15b4ac377d3e85c6b17e403a5ec9e57ee1b3371e543ecf998e09a0
+size 91447948

app.py CHANGED Viewed

@@ -1,8 +1,9 @@
 import os
 import streamlit as st
 import yaml
 import pandas as pd
-from cryptography.fernet import Fernet
 from dotenv import load_dotenv
 from io import StringIO
@@ -22,36 +23,51 @@ def yaml_to_dict(yaml_str):
     return yaml.safe_load(yaml_str)
 def initialize():
     load_dotenv()
     st.session_state.setdefault('model_names', ['SurveyBot3000', 'PsiSent', 'all_mpnet_base_v2'])
     st.session_state.setdefault('loaded_model_name', None)
     st.session_state.setdefault('search_query', None)
     st.session_state.setdefault('db', None)
-    st.session_state.setdefault('results', pd.DataFrame())
     with open('config.yaml', 'r') as stream:
         st.session_state['config'] = yaml.safe_load(stream)
-    if os.environ.get('decrypt_key'):
-        decrypt_key = os.environ.get('decrypt_key')
-        st.session_state.setdefault('decrypt_key', decrypt_key)
     else:
-        st.session_state.setdefault('decrypt_key', None)
 def main():
     st.set_page_config(page_title='Synth-Net')
     st.markdown("# The Synthetic Nomological Net")
-    # st.markdown("#### This is a demo on how to extract trait information from responses to open-ended questions.")
     st.markdown("""
-        Psychological science is experiencing rapid growth in constructs and measures, partly due to refinement and new research areas,
-        but also due to excessive proliferation. This proliferation, driven by academic incentives for novelty, may lead to redundant
         constructs with different names (jangle fallacy) and seemingly similar constructs with little content overlap (jingle fallacy).
         This web application uses state-of-the-art models and methods in natural language processing to search for semantic overlap in measures.
         It analyzes textual data from over 21,000 scales (containing more than 330,000 items) in an effort to reduce redundancies in measures used in the behavioral sciences.
         - 📖 **Preprint (Open Access)**: NA
         - 🖊️ **Cite**: NA
         - 🌐 **Project website**: NA
@@ -61,95 +77,138 @@ def main():
         The web application is maintained by [magnolia psychometrics](https://www.magnolia-psychometrics.com/).
     """, unsafe_allow_html=True)
-    placeholder_demo = st.empty()
-    show_demo(placeholder_demo)
 def show_demo(placeholder):
     with placeholder:
         with st.container():
             st.divider()
-            st.markdown("""
                 ## Try it yourself!
-                Define a scale by entering individual items in YAML format.
                 After form submission, a vector representation for the scale is calculated using the selected encoder model.
                 Cosine similarities between this vector and the representations of existing scales are then computed.
-                The resulting table outputs measures with high semantic overlap.
             """)
-            with st.form("submission_form"):
-                with st.expander(label="Authentication", expanded=True, icon="🔑"):
-                    st.text_input(
-                        label="Encryption key",
-                        value="",
-                        max_chars=None,
-                        key='decrypt_key',
-                        placeholder="A URL-safe base64-encoded 32-byte key"
-                    )
-                with st.expander(label="Model", expanded=False, icon="🧠"):
-                    if st.session_state['loaded_model_name'] is not None:
-                        input_model_index = st.session_state['model_names'].index(st.session_state['input_model_name'])
-                    else:
-                        input_model_index = 0
-                    st.selectbox(
-                        label="Select model",
-                        options=st.session_state['model_names'],
-                        index=input_model_index,
-                        key='input_model_name'
-                    )
-                with st.expander(label="Search Query", expanded=True, icon="🔎"):
-                    if 'input_items' not in st.session_state:
-                        st.session_state['input_items'] = dict_to_yaml(st.session_state['config']['input_items'])
                     st.text_area(
                         label="Search for similar measures by entering items that constitute the scale (YAML-Formatted):",
                         height=175,
                         key='input_items'
                     )
-                submitted = st.form_submit_button(
-                    label="Search Synth-Net",
-                    type="primary",
-                    use_container_width=True
-                )
-                if submitted:
-                    try:
-                        st.session_state['search_query'] = yaml_to_dict(st.session_state['input_items'])
-                    except yaml.YAMLError as e:
-                        st.error(f"Yikes, you better get your YAML straight! Check https://yaml.org/ for help! \n {e}")
-                        return
-                    try:
-                        modeling.load_model()
                         modeling.search()
-                    except Exception as error:
-                        st.error(f"Error while loading model: {error}")
-                        return
-            with st.container():
-                if not st.session_state['results'].empty:
-                    df = st.session_state['results'].style.format({
-                        'Match': '{:.2f}'.format,
-                        'Scale': str.capitalize,
-                        'Instrument': str.capitalize,
-                    })
-                    st.dataframe(df, use_container_width=True, hide_index=True)
-                    st.download_button(
-                        label="Download References",
-                        data=df_to_csv(st.session_state['results']),
-                        file_name='scored_survey_responses.csv',
-                        mime='text/csv',
-                        use_container_width=True
-                    )
 if __name__ == '__main__':
     initialize()

 import os
 import streamlit as st
 import yaml
+import logging
 import pandas as pd
+from cryptography.fernet import Fernet, InvalidToken
 from dotenv import load_dotenv
 from io import StringIO
     return yaml.safe_load(yaml_str)
 def initialize():
+    logging.basicConfig(level=logging.INFO)
     load_dotenv()
     st.session_state.setdefault('model_names', ['SurveyBot3000', 'PsiSent', 'all_mpnet_base_v2'])
+    st.session_state.setdefault('input_model_name', st.session_state['model_names'][0])
     st.session_state.setdefault('loaded_model_name', None)
     st.session_state.setdefault('search_query', None)
     st.session_state.setdefault('db', None)
+    st.session_state.setdefault('search_results', pd.DataFrame())
+    st.session_state.setdefault('explore_plot', None)
+    st.session_state.setdefault('is_authenticated', False)
     with open('config.yaml', 'r') as stream:
         st.session_state['config'] = yaml.safe_load(stream)
+    if os.environ.get('encryption_key'):
+        encryption_key = os.environ.get('encryption_key')
+        st.session_state.setdefault('encryption_key', encryption_key)
+        # st.session_state.setdefault('encryption_key', None)
     else:
+        st.session_state.setdefault('encryption_key', None)
 def main():
     st.set_page_config(page_title='Synth-Net')
     st.markdown("# The Synthetic Nomological Net")
     st.markdown("""
+        Psychological science is experiencing rapid growth in constructs and measures, partly due to refinement and new research areas,
+        but also due to excessive proliferation. This proliferation, driven by academic incentives for novelty, may lead to redundant
         constructs with different names (jangle fallacy) and seemingly similar constructs with little content overlap (jingle fallacy).
         This web application uses state-of-the-art models and methods in natural language processing to search for semantic overlap in measures.
         It analyzes textual data from over 21,000 scales (containing more than 330,000 items) in an effort to reduce redundancies in measures used in the behavioral sciences.
+    """, unsafe_allow_html=True)
+    placeholder_authentication = st.empty()
+    placeholder_demo = st.empty()
+    if st.session_state['is_authenticated']:
+        show_demo(placeholder_demo)
+    else:
+        show_authentication(placeholder_authentication)
+    st.markdown("""
         - 📖 **Preprint (Open Access)**: NA
         - 🖊️ **Cite**: NA
         - 🌐 **Project website**: NA
         The web application is maintained by [magnolia psychometrics](https://www.magnolia-psychometrics.com/).
     """, unsafe_allow_html=True)
+def show_authentication(placeholder):
+    with placeholder:
+        with st.container():
+            with st.form("authentication_form"):
+                st.markdown("""
+                    ## Authentication
+                    This app is a research preview and requires authentication.
+                    All data is encrypted. Please use your 32-byte encryption key to proceed!
+                """)
+                st.text_input(
+                    label="🔑 Encryption key",
+                    value="",
+                    max_chars=None,
+                    key='encryption_key',
+                    placeholder="A URL-safe base64-encoded 32-byte key"
+                )
+                submitted = st.form_submit_button(
+                    label="Authenticate",
+                    type="primary",
+                    use_container_width=True
+                )
+                if submitted:
+                    try:
+                        modeling.load_db()
+                        st.rerun()
+                    except InvalidToken:
+                        error = f"Error: The encryption key you have entered is invalid (**{st.session_state['encryption_key']}**)!"
+                        st.error(body=error, icon="🔑")
+                        logging.error(error)
+                        st.session_state['is_authenticated'] = False
+                        return
+                    except ValueError as error:
+                        st.error(body=error, icon="🔑")
+                        logging.error(error)
+                        st.session_state['is_authenticated'] = False
+                        return
 def show_demo(placeholder):
     with placeholder:
         with st.container():
             st.divider()
+            st.markdown("""
                 ## Try it yourself!
+                Define a scale by entering individual items in YAML format.
                 After form submission, a vector representation for the scale is calculated using the selected encoder model.
                 Cosine similarities between this vector and the representations of existing scales are then computed.
+                The resulting table outputs measures with high semantic overlap.
             """)
+            if st.session_state['loaded_model_name'] is not None:
+                input_model_index = st.session_state['model_names'].index(st.session_state['input_model_name'])
+            else:
+                input_model_index = 0
+            st.selectbox(
+                label="Select model",
+                options=st.session_state['model_names'],
+                index=input_model_index,
+                placeholder="Choose a model",
+                key='input_model_name'
+            )
+            tab1, tab2 = st.tabs(["🔎 Search for scales", "🕸️ Explore the synthetic nomological net"])
+            with tab1:
+                if 'input_items' not in st.session_state:
+                    st.session_state['input_items'] = dict_to_yaml(st.session_state['config']['input_items'])
+                with st.form("submission_form"):
                     st.text_area(
                         label="Search for similar measures by entering items that constitute the scale (YAML-Formatted):",
                         height=175,
                         key='input_items'
                     )
+                    submitted = st.form_submit_button(
+                        label="Search Synth-Net",
+                        type="primary",
+                        use_container_width=True
+                    )
+                    if submitted:
+                        try:
+                            st.session_state['search_query'] = yaml_to_dict(st.session_state['input_items'])
+                        except yaml.YAMLError as e:
+                            st.error(f"Yikes, you better get your YAML straight! Check https://yaml.org/ for help! \n {e}")
+                            return
+                        no_model = st.session_state.get('model') is None
+                        swap_model = st.session_state.get('input_model_name') != st.session_state['loaded_model_name']
+                        if swap_model or no_model:
+                            modeling.load_db()
+                            modeling.load_model()
                         modeling.search()
+                    with st.container():
+                        if not st.session_state['search_results'].empty:
+                            with st.spinner('Rendering search results...'):
+                                df = st.session_state['search_results'].style.format({
+                                    'Match': '{:.2f}'.format,
+                                    'Scale': str.capitalize,
+                                    'Instrument': str.capitalize,
+                                })
+                                st.dataframe(df, use_container_width=True, hide_index=True)
+            with tab2:
+                with st.container():
+                    modeling.explore()
+                    if st.session_state['explore_plot']:
+                        st.plotly_chart(
+                            figure_or_data=st.session_state['explore_plot'],
+                            use_container_width=True
+                        )
+            # if not st.session_state['search_results'].empty:
+            #     st.download_button(
+            #         label="Download References",
+            #         data=df_to_csv(st.session_state['search_results']),
+            #         file_name='scored_survey_responses.csv',
+            #         mime='text/csv',
+            #         use_container_width=True
+            #     )
 if __name__ == '__main__':
     initialize()

modeling.py CHANGED Viewed

@@ -4,72 +4,113 @@ import logging
 import pandas as pd
 import numpy as np
 import pickle
 from cryptography.fernet import Fernet
 from sentence_transformers import SentenceTransformer, util
-def load_model():
-    no_model = st.session_state.get('model') is None
-    swap_model = st.session_state.get('input_model_name') != st.session_state['loaded_model_name']
-    if swap_model or no_model:
-        with st.spinner('Loading the model might take a couple of seconds...'):
-            env_local = st.session_state['input_model_name'].lower() + '_path'
-            env_remote = st.session_state['input_model_name'].lower() + '_remote_path'
-            if os.environ.get(env_remote):
-                model_path = os.environ.get(env_remote)
-            else:
-                model_path = os.getenv(env_local)
-            auth_token = os.environ.get('read_models') or True
-            st.session_state['model'] = SentenceTransformer(
-                model_name_or_path=model_path,
-                use_auth_token=auth_token
-            )
-            st.session_state['loaded_model_name'] = st.session_state['input_model_name']
-            logging.info(f"Loaded {st.session_state['input_model_name']}!")
-        with st.spinner('Loading embeddings...'):
-            file_path = f"./{st.session_state['input_model_name'].lower()}.enc"
-            with open(file_path, 'rb') as f:
-                encrypted_data = f.read()
-            try:
-                cipher = Fernet(st.session_state['decrypt_key'])
-                decrypted_df = cipher.decrypt(encrypted_data)
-                st.session_state['db'] = pickle.loads(decrypted_df)
-            except Exception as e:
-                st.error(body="Error: No valid encryption key!", icon="🔑")
-                logging.error(e)
-                return
-            # st.session_state['db'] = pd.read_parquet(
-            #     path=f"./{st.session_state['input_model_name'].lower()}.parquet"
-            # )
-            #decrypt_key
 def search():
-    with st.spinner('Searching...'):
-        query_embeddings = st.session_state['model'].encode(sentences=st.session_state['search_query']).mean(axis=0)
         query_scores = util.cos_sim(
             a=np.array(query_embeddings),
-            b=st.session_state['db']['ItemStemEmbeddings']
         ).squeeze()
-        st.session_state['results'] = pd.DataFrame({
             'Match': query_scores,
             'Scale': st.session_state['db']['ScaleName'],
             'Instrument': st.session_state['db']['InstrumentName'],
-            'Reference': st.session_state['db']['InstrumentApaReference'],
-        }).sort_values(by='Match', ascending=False)

 import pandas as pd
 import numpy as np
 import pickle
+import numpy as np
+from bertopic import BERTopic
+from umap import UMAP
 from cryptography.fernet import Fernet
 from sentence_transformers import SentenceTransformer, util
+from pdb import set_trace as trace
+def load_db():
+    with st.spinner('Loading pre-computed embeddings...'):
+        if st.session_state['input_model_name']:
+            file_path = f"./{st.session_state['input_model_name'].lower()}.enc"
+        else:
+            file_path = f"./{st.session_state['model_names'][0].lower()}.enc"
+        logging.info(f"Loading data from {file_path}!")
+        with open(file_path, 'rb') as f:
+            encrypted_data = f.read()
+        cipher = Fernet(st.session_state['encryption_key'])
+        decrypted_df = cipher.decrypt(encrypted_data)
+        st.session_state['db'] = pickle.loads(decrypted_df)
+        st.session_state['is_authenticated'] = True
+        logging.info(f"Loaded {file_path}!")
+def load_model():
+    with st.spinner('Loading the model...'):
+        env_local = st.session_state['input_model_name'].lower() + '_path'
+        env_remote = st.session_state['input_model_name'].lower(
+        ) + '_remote_path'
+        if os.environ.get(env_remote):
+            model_path = os.environ.get(env_remote)
+        else:
+            model_path = os.getenv(env_local)
+        logging.info(f"Loading model from {model_path}!")
+        auth_token = os.environ.get('read_models') or True
+        st.session_state['model'] = SentenceTransformer(
+            model_name_or_path=model_path,
+            token=auth_token
+        )
+        st.session_state['loaded_model_name'] = st.session_state['input_model_name']
+        logging.info(f"Loaded {st.session_state['input_model_name']}!")
 def search():
+    with st.spinner('Searching the synthetic net...'):
+        query_embeddings = st.session_state['model'].encode(
+            sentences=st.session_state['search_query']).mean(axis=0)
+        item_embeddings = np.vstack(
+            st.session_state['db']['ItemStemEmbeddings'])
         query_scores = util.cos_sim(
             a=np.array(query_embeddings),
+            b=item_embeddings
         ).squeeze()
+        st.session_state['search_results'] = pd.DataFrame({
             'Match': query_scores,
             'Scale': st.session_state['db']['ScaleName'],
             'Instrument': st.session_state['db']['InstrumentName'],
+            'Reference': st.session_state['db']['psyctest_doi'],
+        }).sort_values(by='Match', ascending=False)
+def explore():
+    df = st.session_state['db']
+    message = f'Modeling synthetic construct space for {df.shape[0]} scales...'
+    logging.info(message)
+    with st.spinner(message):
+        documents = [f'{x}\n{y}' for x, y in zip(
+            df.ScaleName.tolist(), df.InstrumentName.tolist())]
+        embeddings = np.stack(df.ItemStemEmbeddings.to_numpy())
+        topic_model = BERTopic().fit(
+            documents=documents,
+            embeddings=embeddings
+        )
+        reduced_embeddings = UMAP(
+            n_neighbors=10,
+            n_components=2,
+            min_dist=0.0,
+            metric='cosine'
+        ).fit_transform(embeddings)
+        st.session_state['explore_plot'] = topic_model.visualize_documents(
+            docs=documents,
+            reduced_embeddings=reduced_embeddings,
+            hide_annotations=True,
+            hide_document_hover=False,
+            custom_labels=False,
+            title="The Synthetic Nomological Net",
+            width=1500,
+            height=1500
+        )

psisent.enc CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:ed65f3ce71eae48a6f732ca92a3ffea50872c8b348272e94659c1ce74a6d2b82
-size 97110456

 version https://git-lfs.github.com/spec/v1
+oid sha256:3c278cff21708369f5026353cb18d001eb6a4707702f27ef9e410988dc297247
+size 91447948

requirements.txt CHANGED Viewed

@@ -5,4 +5,6 @@ sentence_transformers==2.7.0
 sentencepiece==0.1.99
 altair==4.2.2
 cryptography==41.0.1
 python-dotenv

 sentencepiece==0.1.99
 altair==4.2.2
 cryptography==41.0.1
+matplotlib_venn==1.1.1
+bertopic==0.16.1
 python-dotenv

surveybot3000.enc CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:53dbcafa17705290f4bd04262bf8bbedbee2133b0608004ad03cbc83444c6bda
-size 97110456

 version https://git-lfs.github.com/spec/v1
+oid sha256:b2817f75bf07ae5bba7c02af4494ead8a904847955e0a94bf8cdb4b6fd90004f
+size 91447948