Last active
August 31, 2026 15:01
-
-
Save Dobby233Liu/fea028fbfb2ba12accba87407d219cb7 to your computer and use it in GitHub Desktop.
ev2csv: Exports and slugifies text from RPG Maker MV map data for localization purposes
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Copyright (c) 2026 Liu Wenyuan | |
| # | |
| # BSD Zero Clause License | |
| # | |
| # Permission to use, copy, modify, and/or distribute this software for any | |
| # purpose with or without fee is hereby granted. | |
| # | |
| # THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES | |
| # WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF | |
| # MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR | |
| # ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES | |
| # WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN | |
| # ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF | |
| # OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. | |
| import json | |
| from typing import TypedDict, Self, TypeAlias | |
| from enum import IntEnum | |
| import re | |
| from dataclasses import dataclass | |
| from itertools import chain, batched | |
| import csv | |
| from operator import itemgetter | |
| from sys import stderr | |
| # create ev2csv_config.py with these filled in: | |
| SOURCE_LANG: str | |
| OTHER_LANGS: tuple[str] | |
| CHARA_SYMBOLS: list[str] # 1st entry corresponds to the 1st actor, etc. | |
| from ev2csv_config import SOURCE_LANG, OTHER_LANGS, CHARA_SYMBOLS | |
| ###### event parsing | |
| class EventCommandCode(IntEnum): | |
| ShowText = 101 | |
| TextData = 401 | |
| ShowScrollingText = 105 | |
| ScrollingTextData = 405 | |
| ShowChoices = 102 | |
| ChoiceBranch = 402 | |
| CancelledChoiceBranch = 403 | |
| ConditionalBranch = 111 | |
| Else = 411 | |
| # emitted by the official editor | |
| EndOfBranch = 0 | |
| class EventCommandData(TypedDict): | |
| code: EventCommandCode | |
| indent: int | |
| parameters: list | |
| class EventPageData(TypedDict): | |
| list: list[EventCommandData] | |
| NON_DIALOG = 0 | |
| @dataclass | |
| class LineStackFrame(): | |
| branch: int | |
| line: int | |
| def __init__(self: Self): | |
| super().__init__() | |
| self.clear() | |
| def clear(self: Self): | |
| self.branch = 0 | |
| self.line = 0 | |
| class LineContext(): | |
| stack: list[LineStackFrame] | |
| text: str|None | |
| src_indexes: list[int]|None | |
| def __init__(self: Self): | |
| super().__init__() | |
| self.stack = [] | |
| self.clear() | |
| def clear(self: Self): | |
| if len(self.stack) == 0: | |
| self.stack.append(LineStackFrame()) | |
| else: | |
| self.stack[1:] = [] | |
| self.stack[0].clear() | |
| self.text = None | |
| self.src_indexes = None | |
| def setup_stack(self: Self, depth: int): | |
| orig_depth = len(self.stack) - 1 | |
| if orig_depth > depth: | |
| self.stack[depth+1:] = [] | |
| elif orig_depth < depth: | |
| self.stack.extend(LineStackFrame() for _ in range(depth - orig_depth)) | |
| else: | |
| return | |
| self.text = None | |
| self.src_indexes = None | |
| def parse_cmds(cmds: list[EventCommandData], ctx: LineContext|None = None): | |
| if ctx is None: | |
| ctx = LineContext() | |
| else: | |
| ctx.clear() | |
| text_scratch: None|list[str] = None | |
| text_data_cmd = EventCommandCode.TextData | |
| def finish_text(): | |
| nonlocal text_scratch | |
| ctx.text = "\n".join(text_scratch) | |
| text_scratch = None | |
| last_indent = 0 | |
| ideal_indent_change = 0 | |
| for cmd_index, cmd in enumerate(cmds): | |
| code, params, indent = cmd["code"], cmd["parameters"], cmd["indent"] | |
| if text_scratch is not None and code != text_data_cmd: | |
| finish_text() | |
| yield ctx | |
| ctx.src_indexes = None | |
| indent_change = indent - last_indent | |
| def warn(text): | |
| nonlocal cmd_index | |
| print(f"[!] {text} @ {cmd_index}", file=stderr) | |
| if abs(indent_change) > 1: | |
| warn("indent change > 1") | |
| elif (ideal_indent_change != 0 and indent_change != ideal_indent_change): | |
| warn(f"indent didn't change by {ideal_indent_change}") | |
| last_indent = indent | |
| ideal_indent_change = 0 | |
| # advance the line of the current branch | |
| if code in (EventCommandCode.ShowText, | |
| EventCommandCode.ShowScrollingText): | |
| ctx.setup_stack(indent) | |
| ctx.stack[-1].line += 1 | |
| # preallocate stack for the following branch, | |
| # and advance the line of its parent branch | |
| # (for stuff that result in branching) | |
| elif code in (EventCommandCode.ShowChoices, | |
| EventCommandCode.ConditionalBranch): | |
| ctx.setup_stack(indent + 1) | |
| ctx.stack[-2].line += 1 | |
| # preallocate stack for the following branch | |
| # (for branch prologues) | |
| elif code in (EventCommandCode.ChoiceBranch, | |
| EventCommandCode.CancelledChoiceBranch, | |
| EventCommandCode.Else): | |
| ideal_indent_change = 1 | |
| ctx.setup_stack(indent + 1) | |
| elif code == EventCommandCode.EndOfBranch: | |
| ideal_indent_change = -1 | |
| stacktop = ctx.stack[-1] | |
| match code: | |
| case EventCommandCode.ShowText: | |
| ctx.src_indexes = [] | |
| text_scratch = [] | |
| text_data_cmd = EventCommandCode.TextData | |
| case EventCommandCode.TextData: | |
| if text_data_cmd == EventCommandCode.TextData: | |
| ctx.src_indexes.append(cmd_index) | |
| text_scratch.append(params[0]) | |
| case EventCommandCode.ShowScrollingText: | |
| ctx.src_indexes = [] | |
| text_scratch = [] | |
| text_data_cmd = EventCommandCode.ScrollingTextData | |
| case EventCommandCode.ScrollingTextData: | |
| if text_data_cmd == EventCommandCode.ScrollingTextData: | |
| ctx.src_indexes.append(cmd_index) | |
| text_scratch.append(params[0]) | |
| case EventCommandCode.ShowChoices: | |
| ctx.src_indexes = [cmd_index] | |
| stacktop.line = NON_DIALOG | |
| for branch_no, choice in enumerate(params[0], start=1): | |
| stacktop.branch = branch_no | |
| ctx.text = choice | |
| yield ctx | |
| ctx.src_indexes = None | |
| case EventCommandCode.ChoiceBranch: | |
| stacktop.branch, stacktop.line = 1 + params[0], 0 | |
| case EventCommandCode.CancelledChoiceBranch: | |
| stacktop.branch, stacktop.line = 0, 0 # branch 0 = cancel | |
| case EventCommandCode.ConditionalBranch: | |
| stacktop.branch, stacktop.line = 1, 0 # true | |
| case EventCommandCode.Else: | |
| stacktop.branch, stacktop.line = 2, 0 # false | |
| if text_scratch is not None: | |
| finish_text() | |
| yield ctx | |
| ###### line id assigning logic | |
| # yanfly-style | |
| CHARA_NAMETAG_MAGIC = "\\n<\\an" | |
| CHARA_NAMETAG_MAP = { | |
| f"{CHARA_NAMETAG_MAGIC}[{i}]>": j for i, j in enumerate(CHARA_SYMBOLS, start=1) | |
| } | |
| CHARA_NAMETAG_REGEX = re.compile(CHARA_NAMETAG_MAGIC + r"\[([0-9]+?)\]>") | |
| def get_character_symbol_by_text(text: str) -> str: | |
| if CHARA_NAMETAG_MAGIC not in text: | |
| return None | |
| query = sorted( | |
| ((text.index(i), j) for i, j in CHARA_NAMETAG_MAP.items() if i in text), | |
| key=itemgetter(0)) | |
| if len(query) > 0: | |
| return query[0][1] | |
| nt_match = re.search(CHARA_NAMETAG_REGEX, text) | |
| if nt_match is not None: | |
| actor_id = int(nt_match.group(1)) - 1 | |
| return CHARA_SYMBOLS[actor_id] \ | |
| if actor_id >= 0 and len(CHARA_SYMBOLS) > actor_id \ | |
| else f"Actor{actor_id}" | |
| return None | |
| def assign_line_id(ctx: LineContext) -> str: | |
| parts = [] | |
| # discard the "branch" of the root, and the line no. of stacktop | |
| branch_chain = list(chain.from_iterable( | |
| (i.branch, i.line) for i in ctx.stack))[1:-1] | |
| if len(branch_chain) > 0: | |
| parts.extend(map( | |
| lambda i: ".".join(map(lambda j: str(j), i)), | |
| batched(branch_chain, 2))) | |
| line_no = ctx.stack[-1].line | |
| if line_no != NON_DIALOG: | |
| chara = get_character_symbol_by_text(ctx.text) | |
| if chara is not None: parts.append(chara) | |
| parts.append(str(line_no)) | |
| return "_".join(parts) | |
| ###### csv writing | |
| # gdlocalization-style | |
| CSV_KEY_FIELD = "Key" | |
| CSV_FIELDS = (CSV_KEY_FIELD, "Description", "Comment", SOURCE_LANG, *OTHER_LANGS) | |
| def _write_csv(ctx_iter, select_line_ctx, make_id, name: str, stream): | |
| writer = csv.DictWriter(stream, fieldnames=CSV_FIELDS) | |
| writer.writeheader() | |
| last_line_num, last_branch, last_stack_depth = -1, 0, 1 | |
| should_write_end_row = False | |
| for ctx in ctx_iter: | |
| # region separation stuff | |
| if ctx is None: | |
| last_line_num, last_branch, last_stack_depth = -1, 0, 1 | |
| should_write_end_row = True | |
| continue | |
| if should_write_end_row: | |
| writer.writerow({}) | |
| should_write_end_row = False | |
| line_ctx = select_line_ctx(ctx) | |
| line_num, branch = line_ctx.stack[-1].line, line_ctx.stack[-1].branch | |
| stack_depth = len(line_ctx.stack) | |
| if last_line_num != -1 and ( | |
| (last_line_num != line_num and (line_num == NON_DIALOG | |
| or last_line_num == NON_DIALOG)) | |
| or last_stack_depth > stack_depth | |
| or (line_num != NON_DIALOG and last_branch != branch)): | |
| writer.writerow({}) | |
| last_line_num, last_branch, last_stack_depth = line_num, branch, stack_depth | |
| writer.writerow({ | |
| CSV_KEY_FIELD: make_id(name, ctx), | |
| SOURCE_LANG: line_ctx.text | |
| }) | |
| ###### text patching | |
| def patch_text(ctx: LineContext, key: str, cmds: list[EventCommandData]): | |
| src_indexes = ctx.src_indexes | |
| if src_indexes is None or len(src_indexes) == 0: | |
| return | |
| cmd0 = cmds[src_indexes[0]] | |
| params = cmd0["parameters"] | |
| subst_text = f"{{{{{key}}}}}" | |
| match cmd0["code"]: | |
| case EventCommandCode.TextData \ | |
| | EventCommandCode.ScrollingTextData: | |
| params[0] = subst_text | |
| # insert magic number so that patch_text_cleanup | |
| # can remove these commands later | |
| for cmd in (cmds[i] for i in src_indexes[1:]): | |
| cmd["indent"] = -100 | |
| case EventCommandCode.ShowChoices: | |
| branch_no = ctx.stack[-1].branch - 1 | |
| params[0][branch_no] = subst_text | |
| # this is eh | |
| indent = cmd0["indent"] | |
| for cmd in cmds[src_indexes[0]+1:]: | |
| if cmd["indent"] != indent: continue | |
| sub_code = cmd["code"] | |
| if sub_code == EventCommandCode.ShowChoices: | |
| break | |
| if sub_code != EventCommandCode.ChoiceBranch: | |
| continue | |
| sub_params = cmd["parameters"] | |
| if sub_params[0] == branch_no and sub_params[1] == ctx.text: | |
| sub_params[1] = subst_text | |
| case _: | |
| raise NotImplementedError(cmd0["code"]) | |
| def patch_text_cleanup(cmds: list[EventCommandData]): | |
| return list(filter(lambda i: i["indent"] >= 0, cmds)) | |
| ###### map parsing | |
| class MapEventData(TypedDict): | |
| id: int | |
| name: str | |
| pages: list[EventPageData] | |
| class MapData(TypedDict): | |
| events: list[None|MapEventData] | |
| MapLineContext: TypeAlias = tuple[tuple[str, int], LineContext] | |
| def parse_map(data: MapData, map_id: str, | |
| line_ctx_temporal: LineContext|None = None, | |
| patch: bool = False): | |
| for event in data["events"]: | |
| if event is None: continue | |
| print("Event:", event["name"]) | |
| for page_no, page in enumerate(event["pages"], start=1): | |
| print(f" - Page {page_no}", end="") | |
| ctx = None | |
| cmds = page["list"] | |
| for line_ctx in parse_cmds(cmds, ctx=line_ctx_temporal): | |
| ctx = (event["name"], page_no), line_ctx | |
| yield ctx | |
| if patch: map_patch_text(map_id, ctx, cmds) | |
| if ctx is not None: | |
| if patch: patch_text_cleanup(cmds) | |
| print() | |
| yield None | |
| else: | |
| print(" (no data)") | |
| def map_assign_line_id(map_id: str, ctx: MapLineContext) -> str: | |
| map_ctx, line_ctx = ctx | |
| event, page = map_ctx | |
| return f"{map_id}_{event}_{str(page)}_{assign_line_id(line_ctx)}" | |
| def map_patch_text(map_id: str, ctx: MapLineContext, cmds: list[EventCommandData]): | |
| return patch_text(ctx[1], map_assign_line_id(map_id, ctx), cmds) | |
| def map_write_csv(data: MapData, name: str, stream, patch: bool = False): | |
| line_ctx_temporal = LineContext() | |
| ctx_iter = parse_map(data, name, line_ctx_temporal=line_ctx_temporal, | |
| patch=patch) | |
| _write_csv(ctx_iter, lambda i: i[1], map_assign_line_id, name, stream) | |
| ###### interface | |
| if __name__ == "__main__": | |
| from os import path | |
| from sys import argv | |
| PATCH = True | |
| if len(argv) - 1 < 1: | |
| print("Usage: ev2csv.py Map.json", file=stderr) | |
| print("Writes Map.csv and Map_tl.json to the input directory", file=stderr) | |
| exit(1) | |
| print("Reading input") | |
| inp_path = argv[1] | |
| with open(inp_path, "r", encoding="utf-8") as f: | |
| data = json.load(f) | |
| map_name = path.splitext(path.basename(inp_path))[0] | |
| out_csv_path = path.join(path.dirname(path.abspath(inp_path)), f"{map_name}.csv") | |
| with open(out_csv_path, "w", encoding="utf-8", newline="") as f: | |
| map_write_csv(data, map_name, f, patch=PATCH) | |
| if PATCH: | |
| out_patched_path = path.join(path.dirname(path.abspath(inp_path)), | |
| f"{map_name}_tl.json") | |
| with open(out_patched_path, "w", encoding="utf-8") as f: | |
| json.dump(data, f, indent=2, ensure_ascii=False) |
Author
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
"WISHLIST" (both not necessary for my use case):