Downloads · 30 days
12
39% of all-time downloads
junghan/News_category_segmentation
News_category_segmentation is a machine learning model from junghan. Use it for the machine learning task on the model card, and read the license before you ship it in a product. It is set up for peft.
-------------------------------------------------
Downloads · 30 days
12
39% of all-time downloads
All-time downloads
31
Public
Repo size
710 MB
Likes
0
Public
Click a slice to open those files.
.safetensors698 MB · 98%
From the Hugging Face model README
train_prompt_style = """Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request. Before answering, think carefully about the question and create a step-by-step chain of thoughts to ensure a logical and accurate response.
아래 뉴스를 읽고 '경제', '금리', '외환' 중 하나로 분류하세요.
{}
import os import pandas as pd import torch from transformers import AutoTokenizer, AutoModelForCausalLM from tqdm.auto import tqdm import time
base_dir = ## 설정 test_excel = ## 설정 output_excel = ## 설정
model_id = ## 설정
tokenizer = AutoTokenizer.from_pretrained( model_id, use_fast=True, trust_remote_code=True ) model = AutoModelForCausalLM.from_pretrained( model_id, trust_remote_code=True, torch_dtype=torch.bfloat16, device_map={"": "cuda"}, # 전 파라미터를 GPU로만 배치 # low_cpu_mem_usage=True, # (선택) 메모리 사용을 줄이는 로드 옵션 ) model.config.use_cache = True
inference_prompt_style = """Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request. Before answering, think carefully about the question and create a step-by-step chain of thoughts to ensure a logical and accurate response.
아래 뉴스를 읽고 '경제', '금리', '외환' 중 하나로 분류하세요.
{}
df = pd.read_excel(test_excel, engine='openpyxl') print(f"Loaded {len(df)} examples from {test_excel}")
def predict_label(text: str) -> str: # THEME_HIST(뉴스 본문)만 question에 넣습니다 question = text.strip() # inference_prompt_style에 question만 첫 번째 {}에, 나머지 두 자리는 빈 문자열("")로 채워줍니다 prompt = inference_prompt_style.format(question, "", "") + tokenizer.eos_token
inputs = tokenizer(
prompt,
return_tensors='pt',
truncation=True,
max_length=2048
).to('cuda')
outputs = model.generate(
input_ids=inputs.input_ids,
attention_mask=inputs.attention_mask,
max_new_tokens=100,
eos_token_id=tokenizer.eos_token_id,
use_cache=True,
)
decoded = tokenizer.batch_decode(outputs, skip_special_tokens=True)[0]
# "### Response:" 뒤의 텍스트를 요약으로 가져옵니다
summary = decoded.split("### Response:")[-1].strip()
return summary
import torch from concurrent.futures import ThreadPoolExecutor, as_completed
prompts = [ inference_prompt_style.format(row['THEME_HIST'].strip(), "", "") + tokenizer.eos_token for _, row in df.iterrows() ]
def infer_one(prompt: str) -> str: # LangSmith에 “llm” 타입으로 run 생성 with trace(name="Qwen3-8B Summarization", run_type="llm", inputs={"prompt": prompt}) as run: start = time.time()
# 토크나이징
inputs_tok = tokenizer(
prompt,
return_tensors="pt",
truncation=True,
max_length=2048
).to("cuda")
input_tokens = inputs_tok.input_ids.numel()
# 모델 생성
outputs = model.generate(
input_ids=inputs_tok.input_ids,
attention_mask=inputs_tok.attention_mask,
max_new_tokens=100,
eos_token_id=tokenizer.eos_token_id,
use_cache=True,
)
output_tokens = outputs.sequences.shape[1]
# 디코딩 및 요약 추출
decoded = tokenizer.batch_decode(outputs, skip_special_tokens=True)[0]
summary = decoded.split("### Response:")[-1].strip()
latency_ms = int((time.time() - start) * 1000)
# 메타데이터 기록
run.metadata["input_tokens"] = int(input_tokens)
run.metadata["output_tokens"] = int(output_tokens)
run.metadata["latency_ms"] = latency_ms
# 결과 저장 및 run 종료
run.end(outputs={"summary": summary})
return summary
df['summary'] = ( df['summary'] .astype(str) .str.split(r'</think>', n=1) .str[-1] .str.strip() )