CoolFace
Apppublic

noa151/LeetCodePredictions

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
app.py191 linesDownload Raw Back to root
1import json
2import gradio as gr
3import joblib
4import pandas as pd
5from related_topics_prediction import MultiLabelThresholdOptimizer
6
7
8def convert_to_float(value):
9    if 'K' in value:
10        return float(value.replace('K', '')) * 1_000
11    elif 'M' in value:
12        return float(value.replace('M', '')) * 1_000_000
13    return float(value)  # If it's already a number
14
15
16def convert_to_string(value):
17    if value >= 1_000_000:
18        return f"{value / 1_000_000:.1f}M"
19    elif value >= 1_000:
20        return f"{value / 1_000:.1f}K"
21    return str(int(value))  # Keep it as an integer if it's below 1,000
22
23
24def greet(title, description, difficulty, topics, likes, accepted, submission, comments, is_premium, predict):
25
26    x_new = pd.DataFrame([{
27        'id': 1,
28        'title': str(title),
29        'description': str(description),
30        'is_premium': 1 if is_premium == "premium" else 0,
31        'difficulty': 0 if difficulty == "Easy" else 1 if difficulty == "Hard" else 2,
32        'acceptance_rate': convert_to_float(accepted)/convert_to_float(submission),
33        'frequency': 0,
34        'discuss_count': float(comments),
35        'accepted': convert_to_float(accepted),
36        'submissions': convert_to_float(submission),
37        'companies': [""],
38        'related_topics': topics.split(',') if isinstance(topics, str) else topics,
39        'likes': convert_to_float(likes),
40        'dislikes': 0,
41        'rating': convert_to_float(likes) / (convert_to_float(likes) + 0),
42        'asked_by_faang': 0,
43        'similar_questions': ""
44    }])
45
46    # Efficient Multi-Hot Encoding for Companies
47    company_data = {company: 1 if company in x_new["companies"].iloc[0] else 0 for company in companies_columns}
48    x_new = pd.concat([x_new, pd.DataFrame([company_data])], axis=1)
49
50    x_new = x_new.drop(columns=["companies"])  # Drop original column
51
52    # Efficient Multi-Hot Encoding for Topics
53    topic_data = {topic: 1 if topic in x_new["related_topics"].iloc[0] else 0 for topic in the_topics}
54    x_new = pd.concat([x_new, pd.DataFrame([topic_data])], axis=1)
55
56    x_new = x_new.drop(columns=["related_topics"])  # Drop original topics column
57
58    # Label encode 'title'
59    title_model = joblib.load("title_encoder.pkl")
60    x_new['title'] = title_model.fit_transform(x_new['title'])
61
62    if predict == "related topics":
63        vectorizer = joblib.load("related_topics_vectorizer.pkl")
64
65        new_tfidf = vectorizer.transform(x_new["description"])
66
67        best_model_info = joblib.load('best_model_related_topics_info.pkl')
68        best_model = joblib.load("best_related_topics_model.pkl")
69        optimizer = MultiLabelThresholdOptimizer()
70        optimizer.optimal_thresholds[best_model_info['model_name']] = best_model_info['threshold']
71
72        predictions = optimizer.predict(best_model, new_tfidf, best_model_info['model_name'])
73
74        mlb = joblib.load("related_topics_label_binarizer.pkl")
75        predictions = mlb.inverse_transform(predictions)
76
77        ans = f"the related topics are: {', '.join(map(str, predictions[0]))}"
78        return ans
79
80    else:
81        vectorizer = joblib.load("tfidf_vectorizer.pkl")
82
83        new_tfidf = vectorizer.transform(x_new["description"])
84
85        # Convert to DataFrame
86        new_tfidf_df = pd.DataFrame(new_tfidf.toarray(), columns=vectorizer.get_feature_names_out())
87        x_new = pd.concat([x_new, new_tfidf_df], axis=1)
88        x_new = x_new.drop(columns=['description'])
89
90        if predict == "difficulty level":
91            # load the dislike model because there is no dislike in the input
92            dislikes_model, feature_names = joblib.load("dislikes_XGB_regression_model.pkl")
93
94            x_new_filtered = x_new[feature_names]  # Select only the required features
95            dislike = dislikes_model.predict(x_new_filtered)
96            x_new['dislikes'] = dislike[0]
97            x_new['rating']: convert_to_float(likes) / (convert_to_float(likes) + dislike[0])
98
99            # Load the model
100            class_model = joblib.load("level_classifier_model.pkl")
101
102            # Get feature names from trained model
103            trained_feature_names = class_model.named_steps['standardscaler'].get_feature_names_out()
104
105            x_new = x_new[trained_feature_names]  # Reorder and remove extra columns
106
107            # Fill missing columns with 0 (or a suitable default)
108            for col in trained_feature_names:
109                if col not in x_new:
110                    x_new[col] = 0  # or another default value
111
112            x_new = x_new[trained_feature_names]  # Ensure correct order again
113
114            predictions = class_model.predict(x_new)
115
116            if predictions == 1:
117                prediction = "Hard"
118            elif predictions == 0:
119                prediction = "Easy"
120            elif predictions == 2:
121                prediction = "Medium"
122
123            ans = f"the level difficulty is: {prediction}"
124            return ans
125
126        elif predict == "acceptance":
127            # Load the model
128            accepted_submissions_model, feature_names = joblib.load("accepted_submissions_regression_model.pkl")
129
130            # Assuming `X_new` is a DataFrame with extra features
131            x_new_filtered = x_new[feature_names]  # Select only the required features
132
133            predictions = accepted_submissions_model.predict(x_new_filtered)
134
135            ans = f"the accepted is: {convert_to_string(predictions[0])}"
136            return ans
137
138        elif predict == "number of likes":
139            # Load the model
140            likes_model, feature_names = joblib.load("likes_random_forest_regression_model.pkl")
141
142            # Assuming `X_new` is a DataFrame with extra features
143            x_new_filtered = x_new[feature_names]  # Select only the required features
144
145            predictions = likes_model.predict(x_new_filtered)
146
147            ans = f"the likes amount is: {convert_to_string(predictions[0])}"
148            return ans
149
150        elif predict == "number of dislikes":
151            # Load the model
152            dislikes_model, feature_names = joblib.load("dislikes_XGB_regression_model.pkl")
153
154            # Assuming `x_new` is a DataFrame with extra features
155            x_new_filtered = x_new[feature_names]  # Select only the required features
156
157            predictions = dislikes_model.predict(x_new_filtered)
158
159            ans = f"the dislikes amount is: {convert_to_string(predictions[0])}"
160            return ans
161
162
163with open("encoding_metadata.json", "r") as f:
164    encoding_metadata = json.load(f)
165
166the_topics = encoding_metadata["related_topics_columns"]
167the_topics.remove("")
168companies_columns = encoding_metadata["companies_columns"]
169companies_columns.remove("")
170
171demo = gr.Interface(
172    fn=greet,
173    inputs=[gr.Text(label="Title"), gr.Text(label="Description"),
174            gr.Radio(choices=["Easy", "Medium", "Hard"], label="Difficulty Level"),
175            gr.Dropdown(the_topics, multiselect=True, label="Related Topics",
176                        info="choose all the related topics of this question"),
177            gr.Text(label="Likes Amount"),
178            gr.Text(label="Accepted Amount"),
179            gr.Text(label="Submission Amount"),
180            gr.Text(label="Comments Amount"),
181            gr.Radio(choices=["premium", "not premium"], label="Is Premium"),
182            gr.Radio(choices=["acceptance", "difficulty level", "number of likes", "number of dislikes",
183                              "related topics"], label="Please Predict..")
184            ],
185    outputs=[gr.Text(label="The Prediction")],
186    title="LEETCODE PREDICTOR",
187    description="please go to the leetcode website (https://leetcode.com/problemset/) choose a question and copy the question's detiles to the relevant spaces, then choose what you whould like to predict and submit. the prediction result will appear on the right side of the screen ๐Ÿ˜‰"
188)
189
190demo.launch()
191