SYSUSELab/RustRepoTrans
Evaluating Large Language Models in Repository-level Code Translation RustRepoTrans is the first repository-level code translation benchmark described in the paper "RustRepoTrans: Repository-level Code Translation Benchmark Targeting Rust". Feel free to contact us to submit new results. Benchmark Dataset RustRepoTrans, the first repository-level code translation benchmark comprising 375 tasks targeting Rust, consists of 122 java-rust function pairs, 145 c-rust… See the full description on the dataset page: https://huggingface.co/datasets/SYSUSELab/RustRepoTrans.
3104
1from rank_bm25 import BM25Plus2import os3import sys4import re5from nltk.corpus import stopwords6from nltk.stem import PorterStemmer, WordNetLemmatizer7 8 9def read_corpus(corpus_files_path):10 corpus = []11 12 corpus_files = os.listdir(corpus_files_path)13 for corpus_file in corpus_files:14 with open(os.path.join(corpus_files_path, corpus_file), 'r') as input_file:15 tmp = input_file.read()16 17 corpus.append(tmp)18 19 return corpus20 21def normalize_text(text):22 # 大小写归一化23 text = text.lower()24 25 # 分词26 words = re.findall(r'\w+|[^\s\w]+', text)27 28 # 去除停用词29 stop_words = set(stopwords.words('english'))30 words = [word for word in words if word not in stop_words]31 32 # 词干提取33 stemmer = PorterStemmer()34 words = [stemmer.stem(word) for word in words]35 36 # 词形还原37 lemmatizer = WordNetLemmatizer()38 words = [lemmatizer.lemmatize(word) for word in words]39 40 return words41 42 43# 使用正则表达式进行分词44def tokenize_code(code):45 # 使用归一化46 return normalize_text(code)47 48def main():49 corpus_files_path = "functions"50 query_files_path = "functions_with_unitTest"51 match_results_path = "potential_function_pair"52 53 project = sys.argv[1]54 corpus_lang = sys.argv[2]55 query_lang = sys.argv[3]56 57 corpus_files_path = os.path.join(corpus_files_path, project, corpus_lang)58 query_files_path = os.path.join(query_files_path, project, query_lang)59 match_results_path = os.path.join(match_results_path, project, f"{query_lang}__{corpus_lang}")60 query_files = os.listdir(query_files_path)61 62 # 获取匹配池子63 corpus = read_corpus(corpus_files_path)64 tokenized_corpus = [tokenize_code(doc) for doc in corpus]65 bm25 = BM25Plus(tokenized_corpus)66 67 # 获取请求68 for query_file in query_files:69 with open(os.path.join(query_files_path, query_file), 'r') as input_file:70 query = input_file.read()71 72 # 对于每个请求计算前n个匹配结果73 # 放大函数名的权重74 tokenized_query = tokenize_code(query)75 76 # 获取相关性评分77 scores = bm25.get_scores(tokenized_query)78 # 获取最相关的前几个函数定义79 top_n = 1080 match_results_index = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)[:top_n]81 82 # 如果文件夹不存在,则创建它83 if not os.path.exists(match_results_path):84 os.makedirs(match_results_path)85 86 # 记录匹配结果87 with open(os.path.join(match_results_path, query_file), 'w') as output_file:88 output_file.write("<Target function>\n")89 output_file.write(query)90 output_file.write("\n</Target function>\n\n")91 # for match_result in match_results:92 # output_file.write(match_result)93 # output_file.write("\n")94 output_file.write("<Possible matching functions>\n")95 i = 196 for index in match_results_index:97 output_file.write("<Function {}> \n{}\n</Function {}>\n\n".format(i, corpus[index], i))98 # output_file.write("Score: {}\n".format(scores[index]))99 i += 1100 output_file.write("</Possible matching functions>\n")101 102 103if __name__ == "__main__" :104 main()