CoolFace
Apppublic

isabel/error-testing

sourceHugging Faceupdated 4y agoView on Hugging Face
0likes
app.py141 linesDownload Raw Back to root
1### ----------------------------- ###2###           libraries           ###3### ----------------------------- ###4 5import gradio as gr6import pandas as pd7import numpy as np8from sklearn.model_selection import train_test_split9from sklearn.linear_model import LogisticRegression10from sklearn import metrics11from reader import get_article12 13 14### ------------------------------ ###15###       data transformation      ###16### ------------------------------ ###17 18# load dataset19uncleaned_data = pd.read_csv('data.csv')20 21# remove timestamp from dataset (always first column)22uncleaned_data = uncleaned_data.iloc[: , 1:]23data = pd.DataFrame()24 25# keep track of which columns are categorical and what 26# those columns' value mappings are27# structure: {colname1: {...}, colname2: {...} }28cat_value_dicts = {}29final_colname = uncleaned_data.columns[len(uncleaned_data.columns) - 1]30 31# for each column...32for (colname, colval) in uncleaned_data.iteritems():33 34  # check if col is already a number; if so, add col directly35  # to new dataframe and skip to next column36  if isinstance(colval.values[0], (np.integer, float)):37    data[colname] = uncleaned_data[colname].copy()38    continue39 40  # structure: {0: "lilac", 1: "blue", ...}41  new_dict = {}42  val = 0 # first index per column43  transformed_col_vals = [] # new numeric datapoints44 45  # if not, for each item in that column...46  for (row, item) in enumerate(colval.values):47    48    # if item is not in this col's dict...49    if item not in new_dict:50      new_dict[item] = val51      val += 152    53    # then add numerical value to transformed dataframe54    transformed_col_vals.append(new_dict[item])55  56  # reverse dictionary only for final col (0, 1) => (vals)57  if colname == final_colname:58    new_dict = {value : key for (key, value) in new_dict.items()}59 60  cat_value_dicts[colname] = new_dict61  data[colname] = transformed_col_vals62 63 64### -------------------------------- ###65###           model training         ###66### -------------------------------- ###67 68# select features and predicton; automatically selects last column as prediction69cols = len(data.columns)70num_features = cols - 171x = data.iloc[: , :num_features]72y = data.iloc[: , num_features:]73 74# split data into training and testing sets75x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.25)76 77# instantiate the model (using default parameters)78model = LogisticRegression()79model.fit(x_train, y_train.values.ravel())80y_pred = model.predict(x_test)81 82 83### -------------------------------- ###84###        article generation        ###85### -------------------------------- ###86# borrow file reading function from reader.py87 88def get_feat():89  feats = [abs(x) for x in model.coef_[0]]90  max_val = max(feats)91  idx = feats.index(max_val)92  return data.columns[idx]93  94acc = str(round(metrics.accuracy_score(y_test, y_pred) * 100, 1)) + '%**'95most_imp_feat = get_feat() + "**"96info = get_article(acc, most_imp_feat)97 98 99 100### ------------------------------- ###101###        interface creation       ###102### ------------------------------- ###103 104 105# predictor for generic number of features106def general_predictor(*args):107  features = []108 109  # transform categorical input110  for colname, arg in zip(data.columns, args):111    if (colname in cat_value_dicts):112      features.append(cat_value_dicts[colname][arg])113    else:114      features.append(arg)115 116  # predict single datapoint117  new_input = [features]118  result = model.predict(new_input)119  return cat_value_dicts[final_colname][result[0]]120 121# add data labels to replace those lost via star-args122inputls = []123for colname in data.columns:124  # skip last column125  if colname == final_colname:126    continue127 128  # access categories dict if data is categorical129  # otherwise, just use a number input130  if colname in cat_value_dicts:131    radio_options = list(cat_value_dicts[colname].keys())132    inputls.append(gr.inputs.Radio(choices=radio_options, type="value", label=colname))133  else:134    # add numerical input135    inputls.append(gr.inputs.Number(label=colname))136  137# generate gradio interface138interface = gr.Interface(general_predictor, inputs=inputls, outputs="text", article=info['article'], css=info['css'], theme="grass", title=info['title'], allow_flagging='never', description=info['description'])139 140# show the interface 141interface.launch()