Skip to content

Instantly share code, notes, and snippets.

@Dobby233Liu
Last active August 31, 2026 15:01
Show Gist options
  • Select an option

  • Save Dobby233Liu/fea028fbfb2ba12accba87407d219cb7 to your computer and use it in GitHub Desktop.

Select an option

Save Dobby233Liu/fea028fbfb2ba12accba87407d219cb7 to your computer and use it in GitHub Desktop.
ev2csv: Exports and slugifies text from RPG Maker MV map data for localization purposes
# Copyright (c) 2026 Liu Wenyuan
#
# BSD Zero Clause License
#
# Permission to use, copy, modify, and/or distribute this software for any
# purpose with or without fee is hereby granted.
#
# THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
# WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
# MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
# ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
import json
from typing import TypedDict, Self, TypeAlias
from enum import IntEnum
import re
from dataclasses import dataclass
from itertools import chain, batched
import csv
from operator import itemgetter
from sys import stderr
# create ev2csv_config.py with these filled in:
SOURCE_LANG: str
OTHER_LANGS: tuple[str]
CHARA_SYMBOLS: list[str] # 1st entry corresponds to the 1st actor, etc.
from ev2csv_config import SOURCE_LANG, OTHER_LANGS, CHARA_SYMBOLS
###### event parsing
class EventCommandCode(IntEnum):
ShowText = 101
TextData = 401
ShowScrollingText = 105
ScrollingTextData = 405
ShowChoices = 102
ChoiceBranch = 402
CancelledChoiceBranch = 403
ConditionalBranch = 111
Else = 411
# emitted by the official editor
EndOfBranch = 0
class EventCommandData(TypedDict):
code: EventCommandCode
indent: int
parameters: list
class EventPageData(TypedDict):
list: list[EventCommandData]
NON_DIALOG = 0
@dataclass
class LineStackFrame():
branch: int
line: int
def __init__(self: Self):
super().__init__()
self.clear()
def clear(self: Self):
self.branch = 0
self.line = 0
class LineContext():
stack: list[LineStackFrame]
text: str|None
src_indexes: list[int]|None
def __init__(self: Self):
super().__init__()
self.stack = []
self.clear()
def clear(self: Self):
if len(self.stack) == 0:
self.stack.append(LineStackFrame())
else:
self.stack[1:] = []
self.stack[0].clear()
self.text = None
self.src_indexes = None
def setup_stack(self: Self, depth: int):
orig_depth = len(self.stack) - 1
if orig_depth > depth:
self.stack[depth+1:] = []
elif orig_depth < depth:
self.stack.extend(LineStackFrame() for _ in range(depth - orig_depth))
else:
return
self.text = None
self.src_indexes = None
def parse_cmds(cmds: list[EventCommandData], ctx: LineContext|None = None):
if ctx is None:
ctx = LineContext()
else:
ctx.clear()
text_scratch: None|list[str] = None
text_data_cmd = EventCommandCode.TextData
def finish_text():
nonlocal text_scratch
ctx.text = "\n".join(text_scratch)
text_scratch = None
last_indent = 0
ideal_indent_change = 0
for cmd_index, cmd in enumerate(cmds):
code, params, indent = cmd["code"], cmd["parameters"], cmd["indent"]
if text_scratch is not None and code != text_data_cmd:
finish_text()
yield ctx
ctx.src_indexes = None
indent_change = indent - last_indent
def warn(text):
nonlocal cmd_index
print(f"[!] {text} @ {cmd_index}", file=stderr)
if abs(indent_change) > 1:
warn("indent change > 1")
elif (ideal_indent_change != 0 and indent_change != ideal_indent_change):
warn(f"indent didn't change by {ideal_indent_change}")
last_indent = indent
ideal_indent_change = 0
# advance the line of the current branch
if code in (EventCommandCode.ShowText,
EventCommandCode.ShowScrollingText):
ctx.setup_stack(indent)
ctx.stack[-1].line += 1
# preallocate stack for the following branch,
# and advance the line of its parent branch
# (for stuff that result in branching)
elif code in (EventCommandCode.ShowChoices,
EventCommandCode.ConditionalBranch):
ctx.setup_stack(indent + 1)
ctx.stack[-2].line += 1
# preallocate stack for the following branch
# (for branch prologues)
elif code in (EventCommandCode.ChoiceBranch,
EventCommandCode.CancelledChoiceBranch,
EventCommandCode.Else):
ideal_indent_change = 1
ctx.setup_stack(indent + 1)
elif code == EventCommandCode.EndOfBranch:
ideal_indent_change = -1
stacktop = ctx.stack[-1]
match code:
case EventCommandCode.ShowText:
ctx.src_indexes = []
text_scratch = []
text_data_cmd = EventCommandCode.TextData
case EventCommandCode.TextData:
if text_data_cmd == EventCommandCode.TextData:
ctx.src_indexes.append(cmd_index)
text_scratch.append(params[0])
case EventCommandCode.ShowScrollingText:
ctx.src_indexes = []
text_scratch = []
text_data_cmd = EventCommandCode.ScrollingTextData
case EventCommandCode.ScrollingTextData:
if text_data_cmd == EventCommandCode.ScrollingTextData:
ctx.src_indexes.append(cmd_index)
text_scratch.append(params[0])
case EventCommandCode.ShowChoices:
ctx.src_indexes = [cmd_index]
stacktop.line = NON_DIALOG
for branch_no, choice in enumerate(params[0], start=1):
stacktop.branch = branch_no
ctx.text = choice
yield ctx
ctx.src_indexes = None
case EventCommandCode.ChoiceBranch:
stacktop.branch, stacktop.line = 1 + params[0], 0
case EventCommandCode.CancelledChoiceBranch:
stacktop.branch, stacktop.line = 0, 0 # branch 0 = cancel
case EventCommandCode.ConditionalBranch:
stacktop.branch, stacktop.line = 1, 0 # true
case EventCommandCode.Else:
stacktop.branch, stacktop.line = 2, 0 # false
if text_scratch is not None:
finish_text()
yield ctx
###### line id assigning logic
# yanfly-style
CHARA_NAMETAG_MAGIC = "\\n<\\an"
CHARA_NAMETAG_MAP = {
f"{CHARA_NAMETAG_MAGIC}[{i}]>": j for i, j in enumerate(CHARA_SYMBOLS, start=1)
}
CHARA_NAMETAG_REGEX = re.compile(CHARA_NAMETAG_MAGIC + r"\[([0-9]+?)\]>")
def get_character_symbol_by_text(text: str) -> str:
if CHARA_NAMETAG_MAGIC not in text:
return None
query = sorted(
((text.index(i), j) for i, j in CHARA_NAMETAG_MAP.items() if i in text),
key=itemgetter(0))
if len(query) > 0:
return query[0][1]
nt_match = re.search(CHARA_NAMETAG_REGEX, text)
if nt_match is not None:
actor_id = int(nt_match.group(1)) - 1
return CHARA_SYMBOLS[actor_id] \
if actor_id >= 0 and len(CHARA_SYMBOLS) > actor_id \
else f"Actor{actor_id}"
return None
def assign_line_id(ctx: LineContext) -> str:
parts = []
# discard the "branch" of the root, and the line no. of stacktop
branch_chain = list(chain.from_iterable(
(i.branch, i.line) for i in ctx.stack))[1:-1]
if len(branch_chain) > 0:
parts.extend(map(
lambda i: ".".join(map(lambda j: str(j), i)),
batched(branch_chain, 2)))
line_no = ctx.stack[-1].line
if line_no != NON_DIALOG:
chara = get_character_symbol_by_text(ctx.text)
if chara is not None: parts.append(chara)
parts.append(str(line_no))
return "_".join(parts)
###### csv writing
# gdlocalization-style
CSV_KEY_FIELD = "Key"
CSV_FIELDS = (CSV_KEY_FIELD, "Description", "Comment", SOURCE_LANG, *OTHER_LANGS)
def _write_csv(ctx_iter, select_line_ctx, make_id, name: str, stream):
writer = csv.DictWriter(stream, fieldnames=CSV_FIELDS)
writer.writeheader()
last_line_num, last_branch, last_stack_depth = -1, 0, 1
should_write_end_row = False
for ctx in ctx_iter:
# region separation stuff
if ctx is None:
last_line_num, last_branch, last_stack_depth = -1, 0, 1
should_write_end_row = True
continue
if should_write_end_row:
writer.writerow({})
should_write_end_row = False
line_ctx = select_line_ctx(ctx)
line_num, branch = line_ctx.stack[-1].line, line_ctx.stack[-1].branch
stack_depth = len(line_ctx.stack)
if last_line_num != -1 and (
(last_line_num != line_num and (line_num == NON_DIALOG
or last_line_num == NON_DIALOG))
or last_stack_depth > stack_depth
or (line_num != NON_DIALOG and last_branch != branch)):
writer.writerow({})
last_line_num, last_branch, last_stack_depth = line_num, branch, stack_depth
writer.writerow({
CSV_KEY_FIELD: make_id(name, ctx),
SOURCE_LANG: line_ctx.text
})
###### text patching
def patch_text(ctx: LineContext, key: str, cmds: list[EventCommandData]):
src_indexes = ctx.src_indexes
if src_indexes is None or len(src_indexes) == 0:
return
cmd0 = cmds[src_indexes[0]]
params = cmd0["parameters"]
subst_text = f"{{{{{key}}}}}"
match cmd0["code"]:
case EventCommandCode.TextData \
| EventCommandCode.ScrollingTextData:
params[0] = subst_text
# insert magic number so that patch_text_cleanup
# can remove these commands later
for cmd in (cmds[i] for i in src_indexes[1:]):
cmd["indent"] = -100
case EventCommandCode.ShowChoices:
branch_no = ctx.stack[-1].branch - 1
params[0][branch_no] = subst_text
# this is eh
indent = cmd0["indent"]
for cmd in cmds[src_indexes[0]+1:]:
if cmd["indent"] != indent: continue
sub_code = cmd["code"]
if sub_code == EventCommandCode.ShowChoices:
break
if sub_code != EventCommandCode.ChoiceBranch:
continue
sub_params = cmd["parameters"]
if sub_params[0] == branch_no and sub_params[1] == ctx.text:
sub_params[1] = subst_text
case _:
raise NotImplementedError(cmd0["code"])
def patch_text_cleanup(cmds: list[EventCommandData]):
return list(filter(lambda i: i["indent"] >= 0, cmds))
###### map parsing
class MapEventData(TypedDict):
id: int
name: str
pages: list[EventPageData]
class MapData(TypedDict):
events: list[None|MapEventData]
MapLineContext: TypeAlias = tuple[tuple[str, int], LineContext]
def parse_map(data: MapData, map_id: str,
line_ctx_temporal: LineContext|None = None,
patch: bool = False):
for event in data["events"]:
if event is None: continue
print("Event:", event["name"])
for page_no, page in enumerate(event["pages"], start=1):
print(f" - Page {page_no}", end="")
ctx = None
cmds = page["list"]
for line_ctx in parse_cmds(cmds, ctx=line_ctx_temporal):
ctx = (event["name"], page_no), line_ctx
yield ctx
if patch: map_patch_text(map_id, ctx, cmds)
if ctx is not None:
if patch: patch_text_cleanup(cmds)
print()
yield None
else:
print(" (no data)")
def map_assign_line_id(map_id: str, ctx: MapLineContext) -> str:
map_ctx, line_ctx = ctx
event, page = map_ctx
return f"{map_id}_{event}_{str(page)}_{assign_line_id(line_ctx)}"
def map_patch_text(map_id: str, ctx: MapLineContext, cmds: list[EventCommandData]):
return patch_text(ctx[1], map_assign_line_id(map_id, ctx), cmds)
def map_write_csv(data: MapData, name: str, stream, patch: bool = False):
line_ctx_temporal = LineContext()
ctx_iter = parse_map(data, name, line_ctx_temporal=line_ctx_temporal,
patch=patch)
_write_csv(ctx_iter, lambda i: i[1], map_assign_line_id, name, stream)
###### interface
if __name__ == "__main__":
from os import path
from sys import argv
PATCH = True
if len(argv) - 1 < 1:
print("Usage: ev2csv.py Map.json", file=stderr)
print("Writes Map.csv and Map_tl.json to the input directory", file=stderr)
exit(1)
print("Reading input")
inp_path = argv[1]
with open(inp_path, "r", encoding="utf-8") as f:
data = json.load(f)
map_name = path.splitext(path.basename(inp_path))[0]
out_csv_path = path.join(path.dirname(path.abspath(inp_path)), f"{map_name}.csv")
with open(out_csv_path, "w", encoding="utf-8", newline="") as f:
map_write_csv(data, map_name, f, patch=PATCH)
if PATCH:
out_patched_path = path.join(path.dirname(path.abspath(inp_path)),
f"{map_name}_tl.json")
with open(out_patched_path, "w", encoding="utf-8") as f:
json.dump(data, f, indent=2, ensure_ascii=False)
@Dobby233Liu

Copy link
Copy Markdown
Author

"WISHLIST" (both not necessary for my use case):

  • CommonEvents.json support
  • Fillimg the Comments field with content from Comment commands

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment