# Loads the Universal Sentence Encoder Multilingual module from TensorFlow Hub. base_model_url = "https://tfhub.dev/google/universal-sentence-encoder-multilingual/3" base_model = tf.keras.Sequential([ hub.KerasLayer(base_model_url, input_shape=[], dtype=tf.string, trainable=False) ]) # Defines a list of test sentences. These sentences represent various job titles. test_text = ['Data Scientist', 'Data Analyst', 'Data Engineer', 'Nurse Practitioner', 'Registered Nurse', 'Medical Assistant', 'Social Media Manager', 'Marketing Strategist', 'Product Marketing Manager'] # Creates embeddings for the sentences in the test_text list. # The np.array() function is used to convert the result into a numpy array. # The .tolist() function is used to convert the numpy array into a list, which might be easier to work with. vectors = np.array(base_model.predict(test_text)).tolist() # Calls the plot_similarity function to create a similarity plot. plot_similarity(test_text, vectors, 90, "base model") # Computes STS benchmark score for the base model pearsonr = sts_benchmark(base_model) print("STS Benachmark: " + str(pearsonr))