Team Ai
Datasetpublic

SYSUSELab/RustRepoTrans

Evaluating Large Language Models in Repository-level Code Translation RustRepoTrans is the first repository-level code translation benchmark described in the paper "RustRepoTrans: Repository-level Code Translation Benchmark Targeting Rust". Feel free to contact us to submit new results. Benchmark Dataset RustRepoTrans, the first repository-level code translation benchmark comprising 375 tasks targeting Rust, consists of 122 java-rust function pairs, 145 c-rust… See the full description on the dataset page: https://huggingface.co/datasets/SYSUSELab/RustRepoTrans.

sourceHugging Faceupdated 2y agoView on Hugging Face
3likes104downloads
match_function_throughBm25.py104 linesDownload Raw Back to Dataset_Construction
1from rank_bm25 import BM25Plus2import os3import sys4import re5from nltk.corpus import stopwords6from nltk.stem import PorterStemmer, WordNetLemmatizer7 8 9def read_corpus(corpus_files_path):10    corpus = []11 12    corpus_files = os.listdir(corpus_files_path)13    for corpus_file in corpus_files:14        with open(os.path.join(corpus_files_path, corpus_file), 'r') as input_file:15            tmp = input_file.read()16 17            corpus.append(tmp)18    19    return corpus20 21def normalize_text(text):22    # 大小写归一化23    text = text.lower()24    25    # 分词26    words = re.findall(r'\w+|[^\s\w]+', text)27    28    # 去除停用词29    stop_words = set(stopwords.words('english'))30    words = [word for word in words if word not in stop_words]31    32    # 词干提取33    stemmer = PorterStemmer()34    words = [stemmer.stem(word) for word in words]35    36    # 词形还原37    lemmatizer = WordNetLemmatizer()38    words = [lemmatizer.lemmatize(word) for word in words]39    40    return words41 42 43# 使用正则表达式进行分词44def tokenize_code(code):45    # 使用归一化46    return normalize_text(code)47 48def main():49    corpus_files_path = "functions"50    query_files_path = "functions_with_unitTest"51    match_results_path = "potential_function_pair"52 53    project = sys.argv[1]54    corpus_lang = sys.argv[2]55    query_lang = sys.argv[3]56 57    corpus_files_path = os.path.join(corpus_files_path, project, corpus_lang)58    query_files_path = os.path.join(query_files_path, project, query_lang)59    match_results_path = os.path.join(match_results_path, project, f"{query_lang}__{corpus_lang}")60    query_files = os.listdir(query_files_path)61 62    # 获取匹配池子63    corpus = read_corpus(corpus_files_path)64    tokenized_corpus = [tokenize_code(doc) for doc in corpus]65    bm25 = BM25Plus(tokenized_corpus)66 67    # 获取请求68    for query_file in query_files:69        with open(os.path.join(query_files_path, query_file), 'r') as input_file:70            query = input_file.read()71 72        # 对于每个请求计算前n个匹配结果73        # 放大函数名的权重74        tokenized_query = tokenize_code(query)75        76        # 获取相关性评分77        scores = bm25.get_scores(tokenized_query)78        # 获取最相关的前几个函数定义79        top_n = 1080        match_results_index = sorted(range(len(scores)), key=lambda i: scores[i], reverse=True)[:top_n]81 82        # 如果文件夹不存在,则创建它83        if not os.path.exists(match_results_path):84            os.makedirs(match_results_path)85        86        # 记录匹配结果87        with open(os.path.join(match_results_path, query_file), 'w') as output_file:88            output_file.write("<Target function>\n")89            output_file.write(query)90            output_file.write("\n</Target function>\n\n")91            # for match_result in match_results:92            #     output_file.write(match_result)93            #     output_file.write("\n")94            output_file.write("<Possible matching functions>\n")95            i = 196            for index in match_results_index:97                output_file.write("<Function {}> \n{}\n</Function {}>\n\n".format(i, corpus[index], i))98                # output_file.write("Score: {}\n".format(scores[index]))99                i += 1100            output_file.write("</Possible matching functions>\n")101                102 103if __name__ == "__main__" :104    main()
SYSUSELab/RustRepoTrans · Team Ai