eduardosanchez commited on
Commit
6899d68
·
verified ·
1 Parent(s): ca648da

Update script.py

Browse files
Files changed (1) hide show
  1. script.py +28 -101
script.py CHANGED
@@ -1,111 +1,38 @@
1
- """
2
- IOL-AI 2024 submission — self-contained model script.
 
3
 
4
- Paste this as `script.py` into an empty HF model repo and submit that repo.
5
- It installs its own deps, downloads the model from the Hub, reads the test
6
- set from /tmp/data/test.csv, and writes submission.csv. No weights or
7
- requirements.txt needed in the repo — just this one file.
8
-
9
- Pick a model that fits T4 medium (16 GB VRAM) and finishes 10 rows in < 1 h:
10
-
11
- Thinking (set THINKING = True):
12
- Qwen/Qwen3-14B ~9 GB
13
- Qwen/Qwen3-8B ~5 GB
14
- deepseek-ai/DeepSeek-R1-Distill-Qwen-14B ~9 GB
15
-
16
- Standard (set THINKING = False):
17
- Qwen/Qwen2.5-14B-Instruct ~9 GB
18
- google/gemma-3-12b-it ~8 GB
19
- Qwen/Qwen2.5-7B-Instruct ~5 GB
20
- """
21
- import subprocess
22
- import sys
23
-
24
- subprocess.run([
25
- sys.executable, "-m", "pip", "install", "-q",
26
- "transformers>=4.51", "bitsandbytes>=0.43", "accelerate>=0.30",
27
- "torch>=2.2", "pandas",
28
- ], check=True)
29
-
30
- import re
31
- import torch
32
  import pandas as pd
33
- from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig
 
34
 
35
- # ── Configuration ─────────────────────────────────────────────────────────────
36
- MODEL_ID = "Qwen/Qwen3-14B"
37
- THINKING = True
38
- MAX_NEW_TOKENS = 4096 if THINKING else 512
39
- TEST_PATH = "/tmp/data/test.csv"
40
 
41
- # ── Load model in 4-bit ───────────────────────────────────────────────────────
42
- bnb = BitsAndBytesConfig(
43
- load_in_4bit=True,
44
- bnb_4bit_compute_dtype=torch.bfloat16,
45
- bnb_4bit_use_double_quant=True,
46
- bnb_4bit_quant_type="nf4",
47
- )
48
- tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True)
49
  model = AutoModelForCausalLM.from_pretrained(
50
- MODEL_ID,
51
- quantization_config=bnb,
52
- device_map="auto",
53
- trust_remote_code=True,
54
- )
55
- model.eval()
56
-
57
- # ── Prompt ────────────────────────────────────────────────────────────────────
58
- SYSTEM = (
59
- "You are solving problems from the International Linguistics Olympiad. "
60
- "Each problem gives you linguistic data and asks you to find patterns. "
61
- "Provide one answer per line in the order the items appear. "
62
- "Be concise — output only the answers, no commentary."
63
- )
64
-
65
- def build_prompt(row):
66
- return row["context"].strip() + "\n\n" + row["query"].strip()
67
 
68
- def strip_thinking(text):
69
- return re.sub(r"<think>.*?</think>", "", text, flags=re.DOTALL).strip()
70
 
71
- def run(row):
 
72
  messages = [
73
- {"role": "system", "content": SYSTEM},
74
- {"role": "user", "content": build_prompt(row)},
 
 
75
  ]
76
- kwargs = {"enable_thinking": True} if THINKING else {}
77
- text = tokenizer.apply_chat_template(
78
- messages, tokenize=False, add_generation_prompt=True, **kwargs
79
- )
80
- inputs = tokenizer(text, return_tensors="pt").to(model.device)
81
  with torch.no_grad():
82
- out = model.generate(
83
- **inputs,
84
- max_new_tokens=MAX_NEW_TOKENS,
85
- do_sample=False,
86
- pad_token_id=tokenizer.eos_token_id,
87
- )
88
- gen = out[0][inputs["input_ids"].shape[-1]:]
89
- decoded = tokenizer.decode(gen, skip_special_tokens=True).strip()
90
- return strip_thinking(decoded) if THINKING else decoded
91
-
92
- # ── Run inference ─────────────────────────────────────────────────────────────
93
- test_df = pd.read_csv(TEST_PATH, dtype=str).fillna("")
94
-
95
- preds = []
96
- for i, (_, row) in enumerate(test_df.iterrows()):
97
- try:
98
- pred = run(row)
99
- except Exception as e:
100
- print(f"[{i}] failed: {e}")
101
- pred = ""
102
- preds.append(pred)
103
- print(f" {i+1}/{len(test_df)} done")
104
-
105
- # ── Write submission ──────────────────────────────────────────────────────────
106
- pd.DataFrame({
107
- "id": test_df["id"],
108
- "pred": preds,
109
- "explanation": [""] * len(preds),
110
- }).to_csv("submission.csv", index=False)
111
- print(f"Saved submission.csv — {len(preds)} rows")
 
1
+ import subprocess, sys
2
+ subprocess.run([sys.executable, "-m", "pip", "install", "-q",
3
+ "transformers>=4.43", "accelerate>=0.30", "torch>=2.2", "pandas"], check=True)
4
 
5
+ import json
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6
  import pandas as pd
7
+ import torch
8
+ from transformers import AutoTokenizer, AutoModelForCausalLM
9
 
10
+ MODEL_ID = "Qwen/Qwen2.5-1.5B-Instruct"
 
 
 
 
11
 
12
+ tok = AutoTokenizer.from_pretrained(MODEL_ID)
 
 
 
 
 
 
 
13
  model = AutoModelForCausalLM.from_pretrained(
14
+ MODEL_ID, torch_dtype=torch.float16, device_map="auto"
15
+ ).eval()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
 
17
+ df = pd.read_csv("/tmp/data/test.csv", dtype=str).fillna("")
 
18
 
19
+ rows = []
20
+ for _, r in df.iterrows():
21
  messages = [
22
+ {"role": "system", "content":
23
+ "You solve International Linguistics Olympiad problems. Answer every numbered "
24
+ "item. Put each answer on its own line, in order, with no numbering and no extra text."},
25
+ {"role": "user", "content": f"{r['context'].strip()}\n\n{r['query'].strip()}"},
26
  ]
27
+ ids = tok.apply_chat_template(
28
+ messages, add_generation_prompt=True, return_tensors="pt",
29
+ ).to(model.device)
 
 
30
  with torch.no_grad():
31
+ out = model.generate(ids, max_new_tokens=512, do_sample=False)
32
+ text = tok.decode(out[0][ids.shape[-1]:], skip_special_tokens=True).strip()
33
+ answers = [ln.strip() for ln in text.splitlines() if ln.strip()]
34
+ rows.append({"id": r["id"], "pred": json.dumps(answers, ensure_ascii=False)})
35
+ print(f"{len(rows)}/{len(df)} done", flush=True)
36
+
37
+ pd.DataFrame(rows).to_csv("submission.csv", index=False)
38
+ print("wrote submission.csv", flush=True)