diff --git a/openseek/competition/LongContext-ICL-Annotation/src/main.py b/openseek/competition/LongContext-ICL-Annotation/src/main.py
index c2949795..42197674 100644
--- a/openseek/competition/LongContext-ICL-Annotation/src/main.py
+++ b/openseek/competition/LongContext-ICL-Annotation/src/main.py
@@ -4,20 +4,20 @@
# from method import build_prompt, select_examples, annotate
-from method import build_prompt, select_examples
+from method import build_prompt, select_examples, build_prompt_cot
-from method import annotate_nvidia as annotate # For Nvidia GPU
-# from method import annotate_ascend as annotate # For Huawei Ascend
+# from method import annotate_nvidia as annotate # For Nvidia GPU
+from method import annotate_ascend as annotate # For Huawei Ascend
TASK_FILES = {
- 1: './data/openseek-1_closest_integers.json',
- 2: './data/openseek-2_count_nouns_verbs.json',
- 3: './data/openseek-3_collatz_conjecture.json',
- 4: './data/openseek-4_conala_concat_strings.json',
- 5: './data/openseek-5_semeval_2018_task1_tweet_sadness_detection.json',
- 6: './data/openseek-6_mnli_same_genre_classification.json',
- 7: './data/openseek-7_jeopardy_answer_generation_all.json',
- 8: '../data/openseek-8_kernel_generation.json',
+ 1: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-1_closest_integers.json',
+ 2: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-2_count_nouns_verbs.json',
+ 3: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-3_collatz_conjecture.json',
+ 4: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-4_conala_concat_strings.json',
+ 5: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-5_semeval_2018_task1_tweet_sadness_detection.json',
+ 6: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-6_mnli_same_genre_classification.json',
+ 7: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-7_jeopardy_answer_generation_all.json',
+ 8: '/root/OpenSeek/openseek/competition/LongContext-ICL-Annotation/data/openseek-8_kernel_generation.json',
}
def parser_args():
@@ -30,7 +30,7 @@ def parser_args():
default='../outputs/',
help='Prefix path to save the evaluation logs.')
parser.add_argument('--tokenizer_path', type=str,
- default='/share/project/wuhaiming/spaces/data_agent/OpenSeek-main/openseek/competition/LongContext-ICL-Annotation/src/Qwen3-4B')
+ default='/root/Qwen3-4B')
args = parser.parse_args()
return args
@@ -48,7 +48,7 @@ def evaluate(task_id:int,
task_name = task_dict['task_name']
task_description = task_dict['Definition'][0]
- icl_examples = task_dict['examples'][:100]
+ icl_examples = task_dict['examples'][:50]
test_samples = task_dict['test_samples']
version = 1
@@ -70,7 +70,16 @@ def evaluate(task_id:int,
text2annotate = test_sample['input']
- prompt = build_prompt(task_description, text2annotate)
+
+ # Use CoT prompt for Task 3 and 4, standard prompt for others (Account 3 strategy)
+ # Task 3: Collatz conjecture (math reasoning) - CoT helps
+ # Task 4: String concatenation - CoT significantly helped in Account 2 (+29.2%)
+ # Task 8: Kernel generation - CoT was harmful in Account 2 (6.0% -> 0.6%)
+ if task_id in [3, 4]:
+ prompt = build_prompt_cot(task_description, text2annotate, task_id)
+ else:
+ prompt = build_prompt(task_description, text2annotate)
+
if examples_str is None:
examples_str = select_examples(icl_examples, task_description, text2annotate)
input_prompt = prompt.replace("[[EXAMPLES]]\n\n", examples_str+'\n\n')
@@ -79,9 +88,9 @@ def evaluate(task_id:int,
# if tokenized_input['input_ids'].shape[1] > max_input_length:
# test_record['prediction'] = None
# else:
- # prediction = annotate(input_prompt)
+ # prediction = annotate(input_prompt, task_id)
# test_record['prediction'] = prediction
- prediction = annotate(input_prompt)
+ prediction = annotate(input_prompt, task_id)
test_record['prediction'] = prediction
with open(output_file, 'a') as f:
f.write(json.dumps(test_record)+'\n')
diff --git a/openseek/competition/LongContext-ICL-Annotation/src/method.py b/openseek/competition/LongContext-ICL-Annotation/src/method.py
index 386daf22..f093453b 100644
--- a/openseek/competition/LongContext-ICL-Annotation/src/method.py
+++ b/openseek/competition/LongContext-ICL-Annotation/src/method.py
@@ -49,6 +49,7 @@ def build_prompt____(task_description: str, text2annotate: str) -> str:
def build_prompt(task_description: str, text2annotate: str) -> str:
"""
Construct a high-precision prompt for long-context data annotation (optimized for Qwen3-4B).
+ M01 优化版本:严格标签输出稳态方案
task_description: Clear description of the annotation task (e.g., "Classify English product reviews as Good Review/Bad Review").
text2annotate: The text to be annotated (single text or batch texts).
"""
@@ -60,17 +61,28 @@ def build_prompt(task_description: str, text2annotate: str) -> str:
"### Core Task\n"
f"{task_description}\n\n"
- "### Critical Annotation Guidelines\n"
- "1. **Example Learning Requirement**: Thoroughly analyze and fully learn from the annotation logic, format, and criteria in the Examples section. "
- "Your annotation must align with the style, judgment standards, and tag usage shown in the examples.\n"
- "2. **Thinking Process**: You may (and are encouraged to) explain your annotation reasoning step by step (e.g., key information extraction, judgment basis, rule matching).\n"
- "3. **Mandatory Output Rule**: Regardless of any thinking process you provide, your final annotation result MUST be enclosed in \n"
- "4. **Length Adaptation**: For long texts, maintain complete thinking process and ensure the final tags contain the accurate annotation result (no truncation).\n\n"
+ "### CRITICAL OUTPUT RULES (HIGHEST PRIORITY)\n"
+ "1. **FINAL OUTPUT MANDATE**: Your response MUST contain ONLY the final annotation result wrapped in tags.\n"
+ "2. **STRICTLY PROHIBITED**: ❌ No explanations, reasoning, thinking process, or additional text outside the tags.\n"
+ "3. **MANDATORY FORMAT**: The entire response must be in this exact format: Your_Answer\n"
+ "4. **NO EXCEPTIONS**: Any response that contains text outside tags will be considered INVALID.\n\n"
+
+ "### ERROR EXAMPLES (DO NOT FOLLOW THESE):\n"
+ "❌ WRONG 1: After analysis, I believe this is a positive review. Good Review\n"
+ "❌ WRONG 2: Based on the examples, this text shows: Bad Review\n"
+ "❌ WRONG 3: The answer is: answer\n"
+ "❌ WRONG 4: I think Neutral Review\n"
+ "❌ WRONG 5: Good Review (extra spaces inside tags)\n"
+ "❌ WRONG 6: Good Review (missing tags entirely)\n"
+ "❌ WRONG 7: Bad Review (incomplete tags)\n"
+ "❌ WRONG 8: Reasoning: This text mentions... answer\n\n"
+
+ "### CORRECT EXAMPLES (FOLLOW THESE EXACTLY):\n"
+ "✅ CORRECT 1: Good Review\n"
+ "✅ CORRECT 2: Bad Review\n"
+ "✅ CORRECT 3: Neutral Review\n"
+ "✅ CORRECT 4: 42\n"
+ "✅ CORRECT 5: entailment\n\n"
"### Examples (Must Be Fully Followed)\n"
"[[EXAMPLES]]\n\n"
@@ -78,13 +90,104 @@ def build_prompt(task_description: str, text2annotate: str) -> str:
"### Text to Annotate\n"
f"{text2annotate}\n\n"
- "### Final Requirement Summary\n"
- "1. You can (and should) provide clear thinking process for your annotation.\n"
- "2. The final annotation result MUST be wrapped in tags (no exceptions).\n"
- "3. All annotation logic must strictly follow the examples provided above.\n"
+ "### FINAL COMMAND (READ CAREFULLY):\n"
+ "Your response must be EXACTLY in this format: Your_Answer\n"
+ "No other text, no explanations, no thinking process, no additional content.\n"
+ "Output your answer now: "
)
return prompt
+def build_prompt_cot(task_description: str, text2annotate: str, task_id: int) -> str:
+ """
+ Build a Chain-of-Thought (CoT) prompt for complex reasoning tasks (Task 3, 8).
+ This encourages the model to show step-by-step reasoning before final answer.
+ """
+ if task_id == 3:
+ # Task 3: Collatz Conjecture - Mathematical Reasoning
+ prompt = (
+ "### Role Definition\n"
+ "You are a mathematical reasoning expert specializing in the Collatz conjecture. "
+ "You excel at systematic step-by-step mathematical reasoning and verification.\n\n"
+
+ "### Core Task\n"
+ f"{task_description}\n\n"
+
+ "### Critical Reasoning Guidelines\n"
+ "1. **Step-by-Step Reasoning**: For each input number, you MUST show your complete reasoning process:\n"
+ " - Step 1: Identify the current number\n"
+ " - Step 2: Apply the Collatz rule (if even: n/2; if odd: 3n+1)\n"
+ " - Step 3: Calculate the next number\n"
+ " - Step 4: Continue until reaching 1\n"
+ " - Step 5: Determine the closest integer to 1\n\n"
+
+ "2. **Verification**: Always verify your calculations:\n"
+ " - Check if the rule was applied correctly\n"
+ " - Confirm the sequence reaches 1\n"
+ " - Double-check the final answer\n\n"
+
+ "3. **Output Format**: Your response must follow this structure:\n"
+ " **Reasoning Process:**\n"
+ " [Show your step-by-step calculations here]\n\n"
+ " **Final Answer:** [closest integer]\n\n"
+
+ "### Examples (Must Be Fully Followed)\n"
+ "[[EXAMPLES]]\n\n"
+
+ "### Text to Annotate\n"
+ f"{text2annotate}\n\n"
+
+ "### Final Requirement Summary\n"
+ "1. Show your complete step-by-step reasoning process.\n"
+ "2. Verify each calculation step.\n"
+ "3. Final answer MUST be wrapped in tags.\n"
+ )
+ elif task_id == 8:
+ # Task 8: Kernel Generation - Code Generation
+ prompt = (
+ "### Role Definition\n"
+ "You are an expert programmer specializing in Linux kernel development. "
+ "You excel at writing correct, efficient, and well-structured kernel code.\n\n"
+
+ "### Core Task\n"
+ f"{task_description}\n\n"
+
+ "### Critical Code Generation Guidelines\n"
+ "1. **Step-by-Step Approach**: Before writing code, think through:\n"
+ " - Step 1: Understand the kernel function requirements\n"
+ " - Step 2: Identify necessary kernel APIs and data structures\n"
+ " - Step 3: Design the function structure\n"
+ " - Step 4: Write the code with proper error handling\n"
+ " - Step 5: Review for common kernel coding issues\n\n"
+
+ "2. **Code Quality Requirements**:\n"
+ " - Use correct kernel APIs (e.g., copy_from_user, copy_to_user)\n"
+ " - Handle all error cases properly\n"
+ " - Follow kernel coding style\n"
+ " - Ensure memory safety\n\n"
+
+ "3. **Output Format**: Your response must follow this structure:\n"
+ " **Analysis:**\n"
+ " [Explain your approach and reasoning]\n\n"
+ " **Code:**\n"
+ " [your complete kernel code here]\n\n"
+
+ "### Examples (Must Be Fully Followed)\n"
+ "[[EXAMPLES]]\n\n"
+
+ "### Text to Annotate\n"
+ f"{text2annotate}\n\n"
+
+ "### Final Requirement Summary\n"
+ "1. Analyze the requirements step-by-step.\n"
+ "2. Write correct kernel code with proper error handling.\n"
+ "3. Final code MUST be wrapped in tags.\n"
+ )
+ else:
+ # Fallback to standard prompt for other tasks
+ prompt = build_prompt(task_description, text2annotate)
+
+ return prompt
+
def build_prompt_backup(task_description:str, text2annotate:str)->str:
"""
Construct the prompt for annotation based on the task description.
@@ -158,7 +261,7 @@ def select_examples(all_examples: list[dict], task_description: str, text2annota
"""
# 初始化Qwen3-4B的tokenizer(自动下载/加载千问3-4B的分词器)
# 若本地已下载模型,可替换为本地路径,如 "./qwen3-4b"
- tokenizer = AutoTokenizer.from_pretrained("/share/project/wuhaiming/spaces/data_agent/OpenSeek-main/openseek/competition/LongContext-ICL-Annotation/src/Qwen3-4B", trust_remote_code=True)
+ tokenizer = AutoTokenizer.from_pretrained("/root/Qwen3-4B", trust_remote_code=True)
# 最大上下文长度限制(Qwen3-4B的上下文窗口默认是8k/32k,可根据实际调整)
target_length = 8192 # 若需严格适配Qwen3-4B,建议改为8192(8k)
@@ -204,22 +307,56 @@ def select_examples(all_examples: list[dict], task_description: str, text2annota
def count_answer(text: str) -> tuple[list, dict]:
"""
提取字符串中标签内的所有内容(字符串形式),统计出现次数最多的内容
+ M01 优化版本:增加兜底匹配逻辑
:param text: 包含标签的原始字符串
:return: 出现次数最多的内容列表、所有内容的频次统计字典
"""
+ # 首先尝试标准标签匹配
pattern = r'\s*(.+?)\s*'
content_matches = re.findall(pattern, text, re.DOTALL)
content_counter = Counter(content_matches)
- if not content_counter:
- return None
+ if content_counter:
+ max_count = max(content_counter.values())
+ answer = [content for content, count in content_counter.items() if count == max_count]
+ return answer[0]
+
+ # M01兜底策略1:尝试提取最后一个单词或短语(可能是直接输出的答案)
+ cleaned_text = text.strip()
+ if cleaned_text:
+ # 移除常见的引导词
+ cleaned_text = re.sub(r'^(?:answer:|the answer is:|result:|final:|output:)\s*', '', cleaned_text, flags=re.IGNORECASE)
+ cleaned_text = cleaned_text.strip()
+
+ # 提取最后一行或最后一个句子
+ lines = [line.strip() for line in cleaned_text.split('\n') if line.strip()]
+ if lines:
+ last_line = lines[-1]
+ # 如果最后一行很短(可能是答案),返回它
+ if len(last_line) <= 50:
+ return last_line
+
+ # 否则尝试提取最后一个词组
+ sentences = re.split(r'[.!?;]', last_line)
+ sentences = [s.strip() for s in sentences if s.strip()]
+ if sentences:
+ last_sentence = sentences[-1]
+ if len(last_sentence) <= 50:
+ return last_sentence
- max_count = max(content_counter.values())
- answer = [content for content, count in content_counter.items() if count == max_count]
+ # M01兜底策略2:提取所有独立的单词短语
+ words = re.findall(r'\b[A-Z][a-zA-Z\s]+\b|\b\d+\b', text)
+ if words:
+ # 返回最后一个大写开头的短语或数字
+ for word in reversed(words):
+ if len(word.strip()) >= 2:
+ return word.strip()
- if (len(answer[0]) >= 100):
- return None
- return answer[0]
+ # M01兜底策略3:如果所有方法都失败,返回原始文本的最后100个字符
+ if text:
+ return text.strip()[-100:]
+
+ return None
def annotate_nvidia(input_prompt:str)->list[str]:
@@ -235,7 +372,7 @@ def annotate_nvidia(input_prompt:str)->list[str]:
data = {
"model": "../Qwen3-4B",
"prompt": input_prompt,
- "max_tokens": 10_000, # max_token = 10k
+ "max_tokens": 1024, # max_token = 10k
}
try:
@@ -248,30 +385,63 @@ def annotate_nvidia(input_prompt:str)->list[str]:
prediction = count_answer(whole_result)
return prediction
-def annotate_ascend(input_prompt:str)->list[str]:
+def annotate_ascend(input_prompt:str, task_id:int=None)->list[str]:
"""
Annotate the unlabeled data using an LLM API (Huawei Ascend).
prompts:
A prompt constructed for annotation.
For example, ``["You are a data annotation assistant. Your task is to label ..."]``
+
+ Optimization for Account 3: Differentiated strategy based on task type
+ - Task 3, 4: CoT reasoning with lower temperature (effective for math and string tasks)
+ - Task 8: Standard configuration (CoT harmful for code generation)
+ - Other tasks: Moderate temperature for balanced performance
"""
import openai
openai.api_key = "EMPTY"
openai.base_url = "http://localhost:9010/v1/"
- model = "Qwen3-4B-ascend-flagos"
+ model = "/root/Qwen3-4B"
+
+ # Adjust temperature based on task (Differentiated Strategy)
+ if task_id in [3, 4]:
+ # Lower temperature for CoT reasoning tasks (Task 3: math, Task 4: strings)
+ # This reduces randomness and improves accuracy
+ temperature = 0.3
+ elif task_id == 8:
+ # Standard temperature for code generation (CoT was harmful in Account 2)
+ temperature = 0.7
+ else:
+ # Moderate temperature for other tasks (balanced randomness and accuracy)
+ temperature = 0.5
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": input_prompt}
]
+
+ # Adjust max_tokens based on task
+ if task_id in [3, 4]:
+ # Increased max_tokens for CoT tasks (supports longer reasoning chains)
+ max_tokens = 2048
+ else:
+ # Standard max_tokens for other tasks
+ max_tokens = 1024
+
response = openai.chat.completions.create(
model=model,
messages=messages,
- temperature=0.7,
+ temperature=temperature,
top_p=0.95,
- max_tokens=10_000,
+ max_tokens=max_tokens,
stream=False,
)
whole_result = response.choices[0].message.content
+
+ # Special handling for Task 8 (code generation): return raw model output
+ # Task 8 generates Triton code without tags
+ if task_id == 8:
+ return whole_result.strip()
+
+ # For other tasks, extract label-tagged content
prediction = count_answer(whole_result)
return prediction
diff --git a/openseek/competition/LongContext-ICL-Annotation/src/requirements.txt b/openseek/competition/LongContext-ICL-Annotation/src/requirements.txt
new file mode 100644
index 00000000..87043eae
--- /dev/null
+++ b/openseek/competition/LongContext-ICL-Annotation/src/requirements.txt
@@ -0,0 +1,17 @@
+# Python依赖环境配置
+# 适用于 Huawei Ascend 910C 环境
+
+# 基础依赖
+numpy>=1.24.0
+torch>=2.8.0
+torch_npu>=2.8.0
+torchvision>=0.23.0
+
+# 模型与推理
+transformers>=4.57.0
+vllm_ascend>=0.13.0rc1
+
+# 工具库
+tqdm>=4.67.0
+requests>=2.32.0
+openai>=2.14.0
\ No newline at end of file
diff --git "a/openseek/competition/LongContext-ICL-Annotation/src/\346\212\200\346\234\257\346\212\245\345\221\212-\346\264\245\351\227\250\345\220\210\345\212\233.pdf" "b/openseek/competition/LongContext-ICL-Annotation/src/\346\212\200\346\234\257\346\212\245\345\221\212-\346\264\245\351\227\250\345\220\210\345\212\233.pdf"
new file mode 100644
index 00000000..4d228a91
Binary files /dev/null and "b/openseek/competition/LongContext-ICL-Annotation/src/\346\212\200\346\234\257\346\212\245\345\221\212-\346\264\245\351\227\250\345\220\210\345\212\233.pdf" differ