CoolFace
Modelpublic

junghan/Qwen-3-8B-news-classification

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes14downloads
Model Card

#인퍼런스 코드 예시

import os import pandas as pd import torch from transformers import AutoTokenizer, AutoModelForCausalLM

1) 경로 설정

basedir = '/content/drive/MyDrive/PEFT(Qwen3)' testexcel = os.path.join(basedir, 'TESTSETCLEANED.xlsx') outputexcel = os.path.join(basedir, 'TESTSETCLEANEDFIN_2.xlsx')

2) 허깅페이스 허브 레포 ID (분류용 모델로 변경)

model_id = "junghan/Qwen-3-8B-news-classification"

3) 모델 & 토크나이저 로드

tokenizer = AutoTokenizer.frompretrained( modelid, usefast=True, trustremotecode=True ) model = AutoModelForCausalLM.frompretrained( modelid, trustremotecode=True, torchdtype=torch.bfloat16, devicemap={"": "cuda"}, ) model.config.usecache = True

inferencepromptstyle = """Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request. Before answering, think carefully about the question and create a step-by-step chain of thoughts to ensure a logical and accurate response.

Instruction:

아래 뉴스를 읽고 반드시 '경제', '금리', '외환', 혹은 'None' 중 하나로만 분류하세요. 만약 세 가지 카테고리(경제, 금리, 외환)에 해당하지 않는다면 반드시 'None'으로 답변하세요.

Question:

{question}

Response:

<think> {cot} </think> {category}"""

df = pd.readexcel(testexcel, engine='openpyxl') print(f"Loaded {len(df)} examples from {test_excel}")

def predictlabel(text: str) -> str: # 뉴스 본문(혹은 제목+본문 등)을 question에 넣음 question = text.strip() prompt = inferencepromptstyle.format(question) + tokenizer.eostoken

inputs = tokenizer( prompt, returntensors='pt', truncation=True, maxlength=2048 ).to('cuda') outputs = model.generate( inputids=inputs.inputids, attentionmask=inputs.attentionmask, maxnewtokens=10, # 분류 태스크이므로 10~20이면 충분 eostokenid=tokenizer.eostokenid, usecache=True, ) decoded = tokenizer.batchdecode(outputs, skipspecialtokens=True)[0] # <think> 태그 이후의 첫 줄(정답)만 추출 afterthink = decoded.split("</think>")[-1].strip() # 첫 줄만 정답으로 사용 label = afterthink.splitlines()[0].strip() # 허용된 값만 반환 if label in ["경제", "금리", "외환", ""]: return label # 예외적으로 다른 값이 나오면 빈 문자열 반환 return ""

from concurrent.futures import ThreadPoolExecutor, as_completed from tqdm.auto import tqdm

THEMEHIST 컬럼명에 맞게 수정 (예: 'TITLE'+'CTEXT' 등)

textcol = "THEMEHIST" # 실제 컬럼명에 맞게 수정

results = [] with ThreadPoolExecutor(maxworkers=4) as executor: futures = {executor.submit(predictlabel, row[textcol]): idx for idx, row in df.iterrows()} for future in tqdm(ascompleted(futures), total=len(futures), desc="분류 진행중"): idx = futures[future] try: results.append((idx, future.result())) except Exception as e: results.append((idx, "")) print(f"Error at idx {idx}: {e}")

결과를 원래 순서대로 정렬

results.sort() labels = [label for idx, label in results] df['category'] = labels

df.toexcel(outputexcel, index=False) print(f"분류 결과가 {output_excel}에 저장되었습니다.")

[More Information Needed]

Framework versions

  • PEFT 0.16.0