@misc{2606.08011,author={Lyu, Boxuan and Song, Haiyue and Qu, Zhi and Kamigaito, Hidetaka and Funakoshi, Kotaro and Okumura, Manabu},title={Rewrite to Translate, Translate to Reward: Reinforcement Learning for Source Rewriting in Machine Translation},year={2026},eprint={arXiv:2606.08011}}
2025
arXiv
Minimum Bayes Risk Decoding for Error Span Detection in Reference-Free Automatic Machine Translation Evaluation
@misc{2512.07540,author={Lyu, Boxuan and Song, Haiyue and Kamigaito, Hidetaka and Ding, Chenchen and Tanaka, Hideki and Utiyama, Masao and Funakoshi, Kotaro and Okumura, Manabu},title={Minimum Bayes Risk Decoding for Error Span Detection in Reference-Free Automatic Machine Translation Evaluation},year={2025},eprint={arXiv:2512.07540}}
arXiv
When Alignment Hurts: Decoupling Representational Spaces in Multilingual Models
@misc{2508.12803,author={Elshabrawy, Ahmed and Kaing, Hour and Song, Haiyue and Aji, Alham Fikri and Tanaka, Hideki and Utiyama, Masao and Dabre, Raj},title={When Alignment Hurts: Decoupling Representational Spaces in Multilingual Models},year={2025},eprint={arXiv:2508.12803}}
Journal Articles
2024
JIP
Bilingual Corpus Mining and Multistage Fine-tuning for Improving Machine Translation of Lecture Transcripts
@article{song2024bilingual,pages={628–640},year={2024},author={Song, Haiyue and Dabre, Raj and Chu, Chenhui and Fujita, Atsushi and Kurohashi, Sadao},publisher={Information Processing Society of Japan},journal={Journal of Information Processing},number={0},doi={10.2197/ipsjjip.32.628},url={http://dx.doi.org/10.2197/ipsjjip.32.628},issn={1882-6652},volume={32},title={Bilingual Corpus Mining and Multistage Fine-tuning for Improving Machine Translation of Lecture Transcripts}}
JNLP
DiverSeg: Leveraging Diverse Segmentations with Cross-granularity Alignment for Neural Machine Translation
@article{song2024diverseg,pages={155–188},year={2024},author={Song, Haiyue and Mao, Zhuoyuan and Dabre, Raj and Chu, Chenhui and Kurohashi, Sadao},publisher={Association for Natural Language Processing},journal={Journal of Natural Language Processing},number={1},doi={10.5715/jnlp.31.155},url={http://dx.doi.org/10.5715/jnlp.31.155},issn={2185-8314},volume={31},title={DiverSeg: Leveraging Diverse Segmentations with Cross-granularity Alignment for Neural Machine Translation}}
2023
TALLIP
SelfSeg: A Self-supervised Sub-word Segmentation Method for Neural Machine Translation
@article{song2023selfseg,pages={1–24},month=aug,year={2023},author={Song, Haiyue and Dabre, Raj and Chu, Chenhui and Kurohashi, Sadao and Sumita, Eiichiro},publisher={Association for Computing Machinery (ACM)},journal={ACM Transactions on Asian and Low-Resource Language Information Processing},number={8},doi={10.1145/3610611},url={http://dx.doi.org/10.1145/3610611},issn={2375-4702},volume={22},title={SelfSeg: A Self-supervised Sub-word Segmentation Method for Neural Machine Translation},articleno={215},numpages={24},address={New York, NY, USA}}
JIP
Spatial Hierarchical Attention Network Based Video-guided Machine Translation
@article{gu2023spatial,pages={299–307},year={2023},author={Gu, Weiqi and Song, Haiyue and Chu, Chenhui and Kurohashi, Sadao},publisher={Information Processing Society of Japan},journal={Journal of Information Processing},number={0},doi={10.2197/ipsjjip.31.299},url={http://dx.doi.org/10.2197/ipsjjip.31.299},issn={1882-6652},volume={31},title={Spatial Hierarchical Attention Network Based Video-guided Machine Translation}}
2019
TODAES
Energy-Efficient and Quality-Assured Approximate Computing Framework Using a Co-Training Method
Li Jiang, Zhuoran Song, Haiyue Song, Chengwen Xu, Qiang Xu, Naifeng Jing, Weifeng Zhang, and Xiaoyao Liang
@article{jiang2019energy,pages={1–25},month=aug,year={2019},author={Jiang, Li and Song, Zhuoran and Song, Haiyue and Xu, Chengwen and Xu, Qiang and Jing, Naifeng and Zhang, Weifeng and Liang, Xiaoyao},publisher={Association for Computing Machinery (ACM)},journal={ACM Transactions on Design Automation of Electronic Systems},number={6},doi={10.1145/3342239},url={http://dx.doi.org/10.1145/3342239},issn={1557-7309},volume={24},title={Energy-Efficient and Quality-Assured Approximate Computing Framework Using a Co-Training Method},articleno={59},numpages={25},address={New York, NY, USA}}
International Conferences
2026
EMNLP
OptiMer: Optimal Distribution Vector Merging Is Better than Data Mixing for Continual Pre-Training
@misc{2603.28858,author={Song, Haiyue and Utiyama, Masao},title={OptiMer: Optimal Distribution Vector Merging Is Better than Data Mixing for Continual Pre-Training},year={2026},eprint={arXiv:2603.28858}}
@misc{2605.29897,author={Leiter, Christoph and Song, Haiyue and Kaing, Hour and Tei, Jin and Tanaka, Hideki and Utiyama, Masao and Eger, Steffen},title={ExCAM: Explainable Cultural Awareness Metrics},year={2026},eprint={arXiv:2605.29897}}
EMNLP
Is Human Annotation Necessary? Iterative MBR Distillation for Error Span Detection in Machine Translation
@misc{2603.12983,author={Lyu, Boxuan and Song, Haiyue and Qu, Zhi},title={Is Human Annotation Necessary? Iterative MBR Distillation for Error Span Detection in Machine Translation},year={2026},eprint={arXiv:2603.12983}}
2025
IJCNLP-AACL
Structured Document Translation via Format Reinforcement Learning
@inproceedings{song-etal-2025-structured,title={Structured Document Translation via Format Reinforcement Learning},author={Song, Haiyue and Eschbach-Dymanus, Johannes and Kaing, Hour and Honda, Sumire and Tanaka, Hideki and Buschbeck, Bianka and Utiyama, Masao},editor={Inui, Kentaro and Sakti, Sakriani and Wang, Haofen and Wong, Derek F. and Bhattacharyya, Pushpak and Banerjee, Biplab and Ekbal, Asif and Chakraborty, Tanmoy and Singh, Dhirendra Pratap},booktitle={Proceedings of the 14th International Joint Conference on Natural Language Processing and the 4th Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics},month=dec,year={2025},address={Mumbai, India},publisher={The Asian Federation of Natural Language Processing and The Association for Computational Linguistics},url={https://aclanthology.org/2025.ijcnlp-long.38/},doi={10.18653/v1/2025.ijcnlp-long.38},pages={677--697},isbn={979-8-89176-298-5}}
IJCNLP-AACL
ImageTra: Real-Time Translation for Texts in Image and Video
@inproceedings{kaing-etal-2025-imagetra,title={{I}mage{T}ra: Real-Time Translation for Texts in Image and Video},author={Kaing, Hour and Mao, Jiannan and Song, Haiyue and Ding, Chenchen and Tanaka, Hideki and Utiyama, Masao},editor={Liu, Xuebo and Purwarianti, Ayu},booktitle={Proceedings of The 14th International Joint Conference on Natural Language Processing and The 4th Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics: System Demonstrations},month=dec,year={2025},address={Mumbai, India},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.ijcnlp-demo.1/},doi={10.18653/v1/2025.ijcnlp-demo.1},pages={1--8},isbn={979-8-89176-301-2}}
IJCNLP-AACL
PRALEKHA: Cross-Lingual Document Alignment for Indic Languages
@inproceedings{suryanarayanan-etal-2025-pralekha,title={{PRALEKHA}: Cross-Lingual Document Alignment for {I}ndic Languages},author={Suryanarayanan, Sanjay and Song, Haiyue and Khan, Mohammed Safi Ur Rahman and Kunchukuttan, Anoop and Dabre, Raj},editor={Inui, Kentaro and Sakti, Sakriani and Wang, Haofen and Wong, Derek F. and Bhattacharyya, Pushpak and Banerjee, Biplab and Ekbal, Asif and Chakraborty, Tanmoy and Singh, Dhirendra Pratap},booktitle={Proceedings of the 14th International Joint Conference on Natural Language Processing and the 4th Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics},month=dec,year={2025},address={Mumbai, India},publisher={The Asian Federation of Natural Language Processing and The Association for Computational Linguistics},url={https://aclanthology.org/2025.ijcnlp-long.37/},doi={10.18653/v1/2025.ijcnlp-long.37},pages={662--676},isbn={979-8-89176-298-5}}
IJCNLP-AACL
Multilingual Iterative Model Pruning: What Matters?
@inproceedings{wibowo-etal-2025-multilingual,title={Multilingual Iterative Model Pruning: What Matters?},author={Wibowo, Haryo Akbarianto and Song, Haiyue and Tanaka, Hideki and Utiyama, Masao and Aji, Alham Fikri and Dabre, Raj},editor={Inui, Kentaro and Sakti, Sakriani and Wang, Haofen and Wong, Derek F. and Bhattacharyya, Pushpak and Banerjee, Biplab and Ekbal, Asif and Chakraborty, Tanmoy and Singh, Dhirendra Pratap},booktitle={Proceedings of the 14th International Joint Conference on Natural Language Processing and the 4th Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics},month=dec,year={2025},address={Mumbai, India},publisher={The Asian Federation of Natural Language Processing and The Association for Computational Linguistics},url={https://aclanthology.org/2025.ijcnlp-long.32/},doi={10.18653/v1/2025.ijcnlp-long.32},pages={543--571},isbn={979-8-89176-298-5}}
Emilio Villa-Cueva, Sholpan Bolatzhanova, Diana Turmakhan, Kareem Elzeky, Henok Biadglign Ademtew, Alham Fikri Aji, Vladimir Araujo, Israel Abebe Azime, and 27 more authors
@inproceedings{villa-cueva-etal-2025-cammt,title={{C}a{MMT}: Benchmarking Culturally Aware Multimodal Machine Translation},author={Villa-Cueva, Emilio and Bolatzhanova, Sholpan and Turmakhan, Diana and Elzeky, Kareem and Ademtew, Henok Biadglign and Aji, Alham Fikri and Araujo, Vladimir and Azime, Israel Abebe and Baek, Jinheon and Belcavello, Frederico and Cristobal, Fermin and Cruz, Jan Christian Blaise and Dabre, Mary and Dabre, Raj and Ehsan, Toqeer and Etori, Naome A and Farooqui, Fauzan and Geng, Jiahui and Ivetta, Guido and Jayakumar, Thanmay and Jeong, Soyeong and Lim, Zheng Wei and Mandal, Aishik and Martinelli, Sof{\'i}a and Mihaylov, Mihail Minkov and Orel, Daniil and Pramanick, Aniket and Purkayastha, Sukannya and Salazar, Israfel and Song, Haiyue and Torrent, Tiago Timponi and Yadeta, Debela Desalegn and Hamed, Injy and Tonja, Atnafu Lambebo and Solorio, Thamar},editor={Christodoulopoulos, Christos and Chakraborty, Tanmoy and Rose, Carolyn and Peng, Violet},booktitle={Findings of the Association for Computational Linguistics: EMNLP 2025},month=nov,year={2025},address={Suzhou, China},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.findings-emnlp.1220/},doi={10.18653/v1/2025.findings-emnlp.1220},pages={22423--22441},isbn={979-8-89176-335-7}}
MT Summit
bytF: How Good Are Byte Level N-Gram F-Scores for Automatic Machine Translation Evaluation?
@inproceedings{dabre-etal-2025-bytf,title={byt{F}: How Good Are Byte Level N-Gram {F}-Scores for Automatic Machine Translation Evaluation?},author={Dabre, Raj and Hour, Kaing and Song, Haiyue},editor={Bouillon, Pierrette and Gerlach, Johanna and Girletti, Sabrina and Volkart, Lise and Rubino, Raphael and Sennrich, Rico and Farinha, Ana C. and Gaido, Marco and Daems, Joke and Kenny, Dorothy and Moniz, Helena and Szoc, Sara},booktitle={Proceedings of Machine Translation Summit XX: Volume 1},month=jun,year={2025},address={Geneva, Switzerland},publisher={European Association for Machine Translation},url={https://aclanthology.org/2025.mtsummit-1.29/},pages={378--387},isbn={978-2-9701897-0-1}}
COLING
PrahokBART: A Pre-trained Sequence-to-Sequence Model for Khmer Natural Language Generation
@inproceedings{kaing-etal-2025-prahokbart,title={{P}rahok{BART}: A Pre-trained Sequence-to-Sequence Model for {K}hmer Natural Language Generation},author={Kaing, Hour and Dabre, Raj and Song, Haiyue and Tran, Van-Hien and Tanaka, Hideki and Utiyama, Masao},editor={Rambow, Owen and Wanner, Leo and Apidianaki, Marianna and Al-Khalifa, Hend and Eugenio, Barbara Di and Schockaert, Steven},booktitle={Proceedings of the 31st International Conference on Computational Linguistics},month=jan,year={2025},address={Abu Dhabi, UAE},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.coling-main.87/},pages={1309--1322}}
COLING Workshop
Exploiting Word Sense Disambiguation in Large Language Models for Machine Translation
@inproceedings{tran-etal-2025-exploiting,title={Exploiting Word Sense Disambiguation in Large Language Models for Machine Translation},author={Tran, Van-Hien and Dabre, Raj and Kaing, Hour and Song, Haiyue and Tanaka, Hideki and Utiyama, Masao},editor={Hettiarachchi, Hansi and Ranasinghe, Tharindu and Rayson, Paul and Mitkov, Ruslan and Gaber, Mohamed and Premasiri, Damith and Tan, Fiona Anting and Uyangodage, Lasitha},booktitle={Proceedings of the First Workshop on Language Models for Low-Resource Languages},month=jan,year={2025},address={Abu Dhabi, United Arab Emirates},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.loreslm-1.10/},pages={135--144}}
COLING
Connecting Ideas in ’Lower-Resource’ Scenarios: NLP for National Varieties, Creoles and Other Low-resource Scenarios
@inproceedings{joshi2025connecting,title={Connecting Ideas in 'Lower-Resource' Scenarios: NLP for National Varieties, Creoles and Other Low-resource Scenarios},author={Joshi, Aditya and Kanojia, Diptesh and Lent, Heather and Kaing, Hour and Song, Haiyue},booktitle={Tutorial at the 31st International Conference on Computational Linguistics (COLING 2025)},year={2025},}
@inproceedings{NEURIPS2024_1568882b,year={2024},volume={37},url={https://proceedings.neurips.cc/paper_files/paper/2024/file/1568882ba1a50316e87852542523739c-Paper-Datasets_and_Benchmarks_Track.pdf},title={CVQA: Culturally-diverse Multilingual Visual Question Answering Benchmark},publisher={Curran Associates, Inc.},pages={11479--11505},editor={Globerson, A. and Mackey, L. and Belgrave, D. and Fan, A. and Paquet, U. and Tomczak, J. and Zhang, C.},doi={10.52202/079017-0366},booktitle={Advances in Neural Information Processing Systems},author={Romero, David and Lyu, Chenyang and Wibowo, Haryo Akbarianto and Lynn, Teresa and Hamed, Injy and Kishore, Aditya Nanda and Mandal, Aishik and Dragonetti, Alina and Abzaliev, Artem and Tonja, Atnafu Lambebo and Balcha, Bontu Fufa and Whitehouse, Chenxi and Salamea, Christian and Velasco, Dan John and Adelani, David Ifeoluwa and Le Meur, David and Villa-Cueva, Emilio and Koto, Fajri and Farooqui, Fauzan and Belcavello, Frederico and Batnasan, Ganzorig and Vallejo, Gisela and Caulfield, Grainne and Ivetta, Guido and Song, Haiyue and Ademtew, Henok Biadglign and Maina, Hern\'{a}n and Lovenia, Holy and Azime, Israel Abebe and Cruz, Jan Christian Blaise and Gala, Jay and Geng, Jiahui and Ortiz-Barajas, Jesus-German and Baek, Jinheon and Dunstan, Jocelyn and Alemany, Laura Alonso and Nagasinghe, Kumaranage Ravindu Yasas and Benotti, Luciana and D\textquotesingle Haro, Luis Fernando and Viridiano, Marcelo and Estecha-Garitagoitia, Marcos and Cabrera, Maria Camila Buitrago and Rodr\'{\i}guez-Cantelar, Mario and Jouitteau, M\'{e}lanie and Mihaylov, Mihail and Etori, Naome and Imam, Mohamed Fazli Mohamed and Adilazuarda, Muhammad Farid and Gochoo, Munkhjargal and Otgonbold, Munkh-Erdene and Niyomugisha, Olivier and Silva, Paula M\'{o}nica and Chitale, Pranjal and Dabre, Raj and Chevi, Rendi and Zhang, Ruochen and Diandaru, Ryandito and Cahyawijaya, Samuel and G\'{o}ngora, Santiago and Jeong, Soyeong and Purkayastha, Sukannya and Kuribayashi, Tatsuki and Clifford, Teresa and Jayakumar, Thanmay and Torrent, Tiago Timponi and Ehsan, Toqeer and Araujo, Vladimir and Kementchedjhieva, Yova and Burzo, Zara and Lim, Zheng Wei and Yong, Zheng Xin and Ignat, Oana and Nwatu, Joan and Mihalcea, Rada and Solorio, Thamar and Aji, Alham Fikri}}
AMTA
How Effective is Synthetic Data and Instruction Fine-tuning for Translation with Markup using LLMs?
@inproceedings{dabre-etal-2024-effective,title={How Effective is Synthetic Data and Instruction Fine-tuning for Translation with Markup using {LLM}s?},author={Dabre, Raj and Song, Haiyue and Exel, Miriam and Buschbeck, Bianka and Eschbach-Dymanus, Johannes and Tanaka, Hideki},editor={Knowles, Rebecca and Eriguchi, Akiko and Goel, Shivali},booktitle={Proceedings of the 16th Conference of the Association for Machine Translation in the Americas (Volume 1: Research Track)},month=sep,year={2024},address={Chicago, USA},publisher={Association for Machine Translation in the Americas},url={https://aclanthology.org/2024.amta-research.8/},pages={73--87}}
IWSLT
NICT’s Cascaded and End-To-End Speech Translation Systems using Whisper and IndicTrans2 for the Indic Task
Ranked 1st out of 4 teams in the IWSLT 2024 Indic speech translation track.
@inproceedings{dabre-song-2024-nicts,title={{NICT}{'}s Cascaded and End-To-End Speech Translation Systems using Whisper and {I}ndic{T}rans2 for the {I}ndic Task},author={Dabre, Raj and Song, Haiyue},editor={Salesky, Elizabeth and Federico, Marcello and Carpuat, Marine},booktitle={Proceedings of the 21st International Conference on Spoken Language Translation (IWSLT 2024)},month=aug,year={2024},address={Bangkok, Thailand (in-person and online)},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.iwslt-1.3/},doi={10.18653/v1/2024.iwslt-1.3},pages={17--22}}
EAMT
SubMerge: Merging Equivalent Subword Tokenizations for Subword Regularized Models in Neural Machine Translation
@inproceedings{song-etal-2024-submerge,title={{S}ub{M}erge: Merging Equivalent Subword Tokenizations for Subword Regularized Models in Neural Machine Translation},author={Song, Haiyue and Meyer, Francois and Dabre, Raj and Tanaka, Hideki and Chu, Chenhui and Kurohashi, Sadao},editor={Scarton, Carolina and Prescott, Charlotte and Bayliss, Chris and Oakley, Chris and Wright, Joanna and Wrigley, Stuart and Song, Xingyi and Gow-Smith, Edward and Bawden, Rachel and S{\'a}nchez-Cartagena, V{\'i}ctor M and Cadwell, Patrick and Lapshinova-Koltunski, Ekaterina and Cabarr{\~a}o, Vera and Chatzitheodorou, Konstantinos and Nurminen, Mary and Kanojia, Diptesh and Moniz, Helena},booktitle={Proceedings of the 25th Annual Conference of the European Association for Machine Translation (Volume 1)},month=jun,year={2024},address={Sheffield, UK},publisher={European Association for Machine Translation (EAMT)},url={https://aclanthology.org/2024.eamt-1.15/},pages={147--163}}
@inproceedings{song2024linguistically,title={Linguistically Motivated Neural Machine Translation},author={Song, Haiyue and Kaing, Hour and Dabre, Raj},booktitle={Tutorial at the 25th Annual Conference of the European Association for Machine Translation (EAMT 2024)},year={2024},}
EAMT Workshop
Incorporating Hypernym Features for Improving Low-resource Neural Machine Translation
@inproceedings{chakrabarty-etal-2024-incorporating,title={Incorporating Hypernym Features for Improving Low-resource Neural Machine Translation},author={Chakrabarty, Abhisek and Song, Haiyue and Dabre, Raj and Tanaka, Hideki and Utiyama, Masao},editor={Tezcan, Arda and S{\'a}nchez-Cartagena, V{\'i}ctor M. and Espl{\`a}-Gomis, Miquel},booktitle={Proceedings of the First International Workshop on Knowledge-Enhanced Machine Translation},month=jun,year={2024},address={Sheffield, United Kingdom},publisher={European Association for Machine Translation (EAMT)},url={https://aclanthology.org/2024.kemt-1.1/},pages={1--6}}
LREC-COLING
NGLUEni: Benchmarking and Adapting Pretrained Language Models for Nguni Languages
@inproceedings{meyer-etal-2024-nglueni,title={{NGLUE}ni: Benchmarking and Adapting Pretrained Language Models for Nguni Languages},author={Meyer, Francois and Song, Haiyue and Chakrabarty, Abhisek and Buys, Jan and Dabre, Raj and Tanaka, Hideki},editor={Calzolari, Nicoletta and Kan, Min-Yen and Hoste, Veronique and Lenci, Alessandro and Sakti, Sakriani and Xue, Nianwen},booktitle={Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)},month=may,year={2024},address={Torino, Italia},publisher={ELRA and ICCL},url={https://aclanthology.org/2024.lrec-main.1071/},pages={12247--12258}}
IWSDS
Enhancing Personality Recognition in Dialogue by Data Augmentation and Heterogeneous Conversational Graph Networks
@inproceedings{fu2024enhancing,title={Enhancing Personality Recognition in Dialogue by Data Augmentation and Heterogeneous Conversational Graph Networks},author={Fu, Yahui and Song, Haiyue and Zhao, Tianyu and Kawahara, Tatsuya},booktitle={Proceedings of the 14th International Workshop on Spoken Dialogue Systems Technology (IWSDS 2024)},address={Sapporo, Japan},year={2024},}
2023
EMNLP
GPT-RE: In-context Learning for Relation Extraction using Large Language Models
@inproceedings{wan-etal-2023-gpt,title={{GPT}-{RE}: In-context Learning for Relation Extraction using Large Language Models},author={Wan, Zhen and Cheng, Fei and Mao, Zhuoyuan and Liu, Qianying and Song, Haiyue and Li, Jiwei and Kurohashi, Sadao},editor={Bouamor, Houda and Pino, Juan and Bali, Kalika},booktitle={Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing},month=dec,year={2023},address={Singapore},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2023.emnlp-main.214/},doi={10.18653/v1/2023.emnlp-main.214},pages={3534--3547}}
ACL
Exploring the Impact of Layer Normalization for Zero-shot Neural Machine Translation
@inproceedings{mao-etal-2023-exploring,title={Exploring the Impact of Layer Normalization for Zero-shot Neural Machine Translation},author={Mao, Zhuoyuan and Dabre, Raj and Liu, Qianying and Song, Haiyue and Chu, Chenhui and Kurohashi, Sadao},editor={Rogers, Anna and Boyd-Graber, Jordan and Okazaki, Naoaki},booktitle={Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)},month=jul,year={2023},address={Toronto, Canada},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2023.acl-short.112/},doi={10.18653/v1/2023.acl-short.112},pages={1300--1316}}
EAMT Workshop
Variable-length Neural Interlingua Representations for Zero-shot Neural Machine Translation
@inproceedings{mao-etal-2023-variable,title={Variable-length Neural Interlingua Representations for Zero-shot Neural Machine Translation},author={Mao, Zhuoyuan and Song, Haiyue and Dabre, Raj and Chu, Chenhui and Kurohashi, Sadao},editor={Barreiro, Anabela and Silberztein, Max and Lloret, Elena and Paprzycki, Marcin},booktitle={Proceedings of the 1st International Workshop on Multilingual, Multimodal and Multitask Language Generation},month=jun,year={2023},address={Tampere, Finland},publisher={European Association for Machine Translation},url={https://aclanthology.org/2023.multi3generation-1.3/},pages={16--25}}
EACL
Relation Extraction with Weighted Contrastive Pre-training on Distant Supervision
@inproceedings{wan-etal-2023-relation,title={Relation Extraction with Weighted Contrastive Pre-training on Distant Supervision},author={Wan, Zhen and Cheng, Fei and Liu, Qianying and Mao, Zhuoyuan and Song, Haiyue and Kurohashi, Sadao},editor={Vlachos, Andreas and Augenstein, Isabelle},booktitle={Findings of the Association for Computational Linguistics: EACL 2023},month=may,year={2023},address={Dubrovnik, Croatia},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2023.findings-eacl.195/},doi={10.18653/v1/2023.findings-eacl.195},pages={2580--2585}}
2022
AACL
BERTSeg: BERT Based Unsupervised Subword Segmentation for Neural Machine Translation
@inproceedings{song-etal-2022-bertseg,title={{BERTS}eg: {BERT} Based Unsupervised Subword Segmentation for Neural Machine Translation},author={Song, Haiyue and Dabre, Raj and Mao, Zhuoyuan and Chu, Chenhui and Kurohashi, Sadao},editor={He, Yulan and Ji, Heng and Li, Sujian and Liu, Yang and Chang, Chua-Hui},booktitle={Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing (Volume 2: Short Papers)},month=nov,year={2022},address={Online only},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2022.aacl-short.12/},doi={10.18653/v1/2022.aacl-short.12},pages={85--94}}
NAACL
When do Contrastive Word Alignments Improve Many-to-many Neural Machine Translation?
@inproceedings{mao-etal-2022-contrastive,title={When do Contrastive Word Alignments Improve Many-to-many Neural Machine Translation?},author={Mao, Zhuoyuan and Chu, Chenhui and Dabre, Raj and Song, Haiyue and Wan, Zhen and Kurohashi, Sadao},editor={Carpuat, Marine and de Marneffe, Marie-Catherine and Meza Ruiz, Ivan Vladimir},booktitle={Findings of the Association for Computational Linguistics: NAACL 2022},month=jul,year={2022},address={Seattle, United States},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2022.findings-naacl.134/},doi={10.18653/v1/2022.findings-naacl.134},pages={1766--1775}}
2021
ACL SRW
Video-guided Machine Translation with Spatial Hierarchical Attention Network
@inproceedings{gu-etal-2021-video,title={Video-guided Machine Translation with Spatial Hierarchical Attention Network},author={Gu, Weiqi and Song, Haiyue and Chu, Chenhui and Kurohashi, Sadao},editor={Kabbara, Jad and Lin, Haitao and Paullada, Amandalynne and Vamvas, Jannis},booktitle={Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing: Student Research Workshop},month=aug,year={2021},address={Online},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2021.acl-srw.9/},doi={10.18653/v1/2021.acl-srw.9},pages={87--92}}
2020
EMNLP Workshop
A System for Worldwide COVID-19 Information Aggregation
Akiko Aizawa, Frederic Bergeron, Junjie Chen, Fei Cheng, Katsuhiko Hayashi, Kentaro Inui, Hiroyoshi Ito, Daisuke Kawahara, and 21 more authors
EMNLP 2020 Workshop, also at the Workshop on NLP for COVID-19 at ACL 2020
@inproceedings{aizawa-etal-2020-system,title={A System for Worldwide {COVID}-19 Information Aggregation},author={Aizawa, Akiko and Bergeron, Frederic and Chen, Junjie and Cheng, Fei and Hayashi, Katsuhiko and Inui, Kentaro and Ito, Hiroyoshi and Kawahara, Daisuke and Kitsuregawa, Masaru and Kiyomaru, Hirokazu and Kobayashi, Masaki and Kodama, Takashi and Kurohashi, Sadao and Liu, Qianying and Matsubara, Masaki and Miyao, Yusuke and Morishima, Atsuyuki and Murawaki, Yugo and Omura, Kazumasa and Song, Haiyue and Sumita, Eiichiro and Suzuki, Shinji and Tanaka, Ribeka and Tanaka, Yu and Toyoda, Masashi and Ueda, Nobuhiro and Ueoka, Honai and Utiyama, Masao and Zhong, Ying},editor={Verspoor, Karin and Cohen, Kevin Bretonnel and Conway, Michael and de Bruijn, Berry and Dredze, Mark and Mihalcea, Rada and Wallace, Byron},booktitle={Proceedings of the 1st Workshop on {NLP} for {COVID}-19 (Part 2) at {EMNLP} 2020},month=dec,year={2020},address={Online},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2020.nlpcovid19-2.13/},doi={10.18653/v1/2020.nlpcovid19-2.13}}
ACL SRW
Pre-training via Leveraging Assisting Languages for Neural Machine Translation
@inproceedings{song-etal-2020-pre,title={Pre-training via Leveraging Assisting Languages for Neural Machine Translation},author={Song, Haiyue and Dabre, Raj and Mao, Zhuoyuan and Cheng, Fei and Kurohashi, Sadao and Sumita, Eiichiro},editor={Rijhwani, Shruti and Liu, Jiangming and Wang, Yizhong and Dror, Rotem},booktitle={Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: Student Research Workshop},month=jul,year={2020},address={Online},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2020.acl-srw.37/},doi={10.18653/v1/2020.acl-srw.37},pages={279--285}}
LREC
Coursera Corpus Mining and Multistage Fine-Tuning for Improving Lectures Translation
@inproceedings{song-etal-2020-coursera,title={{C}oursera Corpus Mining and Multistage Fine-Tuning for Improving Lectures Translation},author={Song, Haiyue and Dabre, Raj and Fujita, Atsushi and Kurohashi, Sadao},editor={Calzolari, Nicoletta and B{\'e}chet, Fr{\'e}d{\'e}ric and Blache, Philippe and Choukri, Khalid and Cieri, Christopher and Declerck, Thierry and Goggi, Sara and Isahara, Hitoshi and Maegaard, Bente and Mariani, Joseph and Mazo, H{\'e}l{\`e}ne and Moreno, Asuncion and Odijk, Jan and Piperidis, Stelios},booktitle={Proceedings of the Twelfth Language Resources and Evaluation Conference},month=may,year={2020},address={Marseille, France},publisher={European Language Resources Association},url={https://aclanthology.org/2020.lrec-1.449/},pages={3640--3649},language={eng},isbn={979-10-95546-34-4}}
LREC
JASS: Japanese-specific Sequence to Sequence Pre-training for Neural Machine Translation
@inproceedings{mao-etal-2020-jass,title={{JASS}: {J}apanese-specific Sequence to Sequence Pre-training for Neural Machine Translation},author={Mao, Zhuoyuan and Cromieres, Fabien and Dabre, Raj and Song, Haiyue and Kurohashi, Sadao},editor={Calzolari, Nicoletta and B{\'e}chet, Fr{\'e}d{\'e}ric and Blache, Philippe and Choukri, Khalid and Cieri, Christopher and Declerck, Thierry and Goggi, Sara and Isahara, Hitoshi and Maegaard, Bente and Mariani, Joseph and Mazo, H{\'e}l{\`e}ne and Moreno, Asuncion and Odijk, Jan and Piperidis, Stelios},booktitle={Proceedings of the Twelfth Language Resources and Evaluation Conference},month=may,year={2020},address={Marseille, France},publisher={European Language Resources Association},url={https://aclanthology.org/2020.lrec-1.454/},pages={3683--3691},language={eng},isbn={979-10-95546-34-4}}
2018
ICCAD
Invocation-driven neural approximate computing with a multiclass-classifier and multiple approximators
Haiyue Song, Chengwen Xu, Qiang Xu, Zhuoran Song, Naifeng Jing, Xiaoyao Liang, and Li Jiang
@inproceedings{song2018invocation,collection={ICCAD ’18},pages={1–8},month=nov,year={2018},author={Song, Haiyue and Xu, Chengwen and Xu, Qiang and Song, Zhuoran and Jing, Naifeng and Liang, Xiaoyao and Jiang, Li},publisher={ACM},booktitle={Proceedings of the International Conference on Computer-Aided Design},doi={10.1145/3240765.3240819},url={http://dx.doi.org/10.1145/3240765.3240819},title={Invocation-driven neural approximate computing with a multiclass-classifier and multiple approximators},series={ICCAD ’18},editor={Bahar, Iris},address={San Diego, CA, USA}}
FPGA
A FPGA Friendly Approximate Computing Framework with Hybrid Neural Networks: (Abstract Only)
Haiyue Song, Xiang Song, Tianjian Li, Hao Dong, Naifeng Jing, Xiaoyao Liang, and Li Jiang
@inproceedings{song2018fpga,collection={FPGA ’18},pages={286–286},month=feb,year={2018},author={Song, Haiyue and Song, Xiang and Li, Tianjian and Dong, Hao and Jing, Naifeng and Liang, Xiaoyao and Jiang, Li},publisher={ACM},booktitle={Proceedings of the 2018 ACM/SIGDA International Symposium on Field-Programmable Gate Arrays},doi={10.1145/3174243.3174965},url={http://dx.doi.org/10.1145/3174243.3174965},title={A FPGA Friendly Approximate Computing Framework with Hybrid Neural Networks: (Abstract Only)},series={FPGA ’18}}
Domestic Conferences (non peer-reviewed)
2026
ANLP
FormatRL: Format Reinforcement Learning for Structured Document Translation
宋 海越, Johannes Eschbach-Dymanus, Hour Kaing, Sumire Honda, 田中 英輝, Bianka Buschbeck, and 内山 将夫
@inproceedings{song2026formatrlnlp,title={FormatRL: Format Reinforcement Learning for Structured Document Translation},author={{宋 海越} and {Johannes Eschbach-Dymanus} and {Hour Kaing} and {Sumire Honda} and {田中 英輝} and {Bianka Buschbeck} and {内山 将夫}},booktitle={言語処理学会第32回年次大会(NLP2026)},address={宇都宮},month=mar,year={2026}}
ANLP
Profanity as a Cue: When LLMs Mistake Toxicity for Hate
Haotian Ye, 宋 海越, Hour Kaing, 丁 塵辰, 田中 英輝, and 内山 将夫
@inproceedings{ye2026profanity,title={Profanity as a Cue: When LLMs Mistake Toxicity for Hate},author={{Haotian Ye} and {宋 海越} and {Hour Kaing} and {丁 塵辰} and {田中 英輝} and {内山 将夫}},booktitle={言語処理学会第32回年次大会(NLP2026)},address={宇都宮},month=mar,year={2026}}
@inproceedings{lyu2025yans,title={生成型自動評価指標のための最小ベイズリスク復号},author={{呂 博軒} and {宋 海越} and {上垣外 英剛} and {田中 英輝} and {内山 将夫} and {船越 孝太郎} and {奥村 学}},booktitle={第20回言語処理若手シンポジウム(YANS2025)},year={2025}}
ANLP
Towards Scene Text Translation for Complex Writing Systems
@inproceedings{kaing2025scenetext,title={Towards Scene Text Translation for Complex Writing Systems},author={{Hour Kaing} and {宋 海越} and {丁 塵辰} and {毛 剣楠} and {田中 英輝} and {内山 将夫}},booktitle={言語処理学会第31回年次大会(NLP2025)},address={長崎},month=mar,year={2025},}
2024
ANLP
Robust Neural Machine Translation for Abugidas by Glyph Perturbation
@inproceedings{kaing2024robust,title={Robust Neural Machine Translation for Abugidas by Glyph Perturbation},author={Kaing, Hour and Ding, Chenchen and Song, Haiyue and Mao, Jiannan and Tanaka, Hideki and Utiyama, Masao},booktitle={言語処理学会第30回年次大会(NLP2024)},address={神戸},month=mar,year={2024}}
2023
ANLP
Large Pre-trained Language Models with Multilingual Prompt for Japanese Natural Language Tasks
@inproceedings{song2023prompt,title={Large Pre-trained Language Models with Multilingual Prompt for Japanese Natural Language Tasks},author={Song, Haiyue and Dabre, Raj and Chu, Chenhui and Kurohashi, Sadao},booktitle={言語処理学会第29回年次大会(NLP2023)},address={沖縄},month=mar,year={2023}}
2022
ANLP
Representative Data Selection for Sequence-to-Sequence Pre-training
@inproceedings{song2022representative,title={Representative Data Selection for Sequence-to-Sequence Pre-training},author={Song, Haiyue and Dabre, Raj and Mao, Zhuoyuan and Chu, Chenhui and Kurohashi, Sadao},booktitle={言語処理学会第28回年次大会(NLP2022)},pages={1--5},month=mar,year={2022}}
ANLP
Improving Medical Relation Extraction with Distantly Supervised Pre-training
@inproceedings{wan2022improving,title={Improving Medical Relation Extraction with Distantly Supervised Pre-training},author={Wan, Zhen and Cheng, Fei and Mao, Zhuoyuan and Liu, Qianying and Song, Haiyue and Kurohashi, Sadao},booktitle={言語処理学会第28回年次大会(NLP2022)},address={浜松},month=mar,year={2022}}
2021
ANLP
Self-supervised Dynamic Programming Encoding for Neural Machine Translation
@inproceedings{song2021selfsupervised,title={Self-supervised Dynamic Programming Encoding for Neural Machine Translation},author={Song, Haiyue and Dabre, Raj and Chu, Chenhui and Kurohashi, Sadao and Sumita, Eiichiro},booktitle={言語処理学会第27回年次大会(NLP2021)},address={北九州},month=mar,year={2021}}
ANLP
Video-guided Machine Translation with Spatial Hierarchical Attention Network Encoder
@inproceedings{gu2021videoguided,title={Video-guided Machine Translation with Spatial Hierarchical Attention Network Encoder},author={Gu, Weiqi and Song, Haiyue and Chu, Chenhui and Kurohashi, Sadao},booktitle={言語処理学会第27回年次大会(NLP2021)},address={北九州},month=mar,year={2021}}
2020
ANLP
Domain Adaptation of Neural Machine Translation through Multistage Fine-Tuning
@inproceedings{song2020domain,title={Domain Adaptation of Neural Machine Translation through Multistage Fine-Tuning},author={Song, Haiyue and Dabre, Raj and Fujita, Atsushi and Kurohashi, Sadao},booktitle={言語処理学会第26回年次大会(NLP2020)},address={茨城},pages={461--464},month=mar,year={2020},}
@inproceedings{mao2020multitask,title={ニューラル機械翻訳のための言語知識に基づくマルチタスク事前学習},author={{Zhuoyuan Mao} and {Raj Dabre} and {Fabien Cromieres} and {宋 海越} and {中尾 亮太} and {黒橋 禎夫}},booktitle={言語処理学会第26回年次大会(NLP2020)},address={茨城},pages={1061--1064},month=mar,year={2020},}