Compare commits

2 Commits

Author SHA1 Message Date
a4e0ae0b08 add --fix-xml
압축파일 내의 경로를 제거하도록 util.py 수정
2026-08-02 19:54:11 +09:00
178776d307 & 특수문자에 amp; 처리가 되어 있지 않은 경우 --fix-xml 옵션을 통해 수정할 수 있도록 추가 2026-07-07 18:11:17 +09:00
2 changed files with 76 additions and 15 deletions

View File

@@ -232,10 +232,9 @@ def ExtractZIP(zip_file: str, extract_to: str):
#
def CreateZIP(output_zip: str, files: list[str]) -> bool:
with zipfile.ZipFile(output_zip, 'w') as zf:
with zipfile.ZipFile(output_zip, 'w', compression=zipfile.ZIP_DEFLATED) as zf:
for file in files:
pathTemp = os.path.join('root', os.path.basename(file))
zf.write(file, pathTemp)
zf.write(file, os.path.basename(file))
bRet = False
if os.path.exists(output_zip):

86
main.py
View File

@@ -40,6 +40,7 @@ def main(argc, argv):
parser.add_argument('--metadata', dest='bMetadata', action='store_true', help='.metadata 파일로부터 ComicInfo.xml 생성 및 CBZ에 추가')
parser.add_argument('--verbose', dest='bVerbose', action='store_true', help='상세 로그 출력')
parser.add_argument('--fix-arc-path', dest='bFixArcPath', action='store_true', help='CBZ 아카이브의 내부 폴더 구조를 제거합니다.')
parser.add_argument('--fix-xml', dest='bFixXml', action='store_true', help='ComicInfo.xml의 특수문자(&) 이스케이프 및 문법 오류를 검사하고 수정합니다.')
parser.add_argument('-v', '--version', action='version', version='ComicInfoXMLConv 1.0')
args = parser.parse_args(argv[1:]) # argv[0]은 스크립트 이름 제외
@@ -92,10 +93,13 @@ def process_archive_file(archive_path: str, args: argparse.Namespace):
arc_number = ExpandPathVars(args.info_number, archive_path)
arc_tags = args.info_tags.split(',') if args.info_tags else []
# info_count를 int로 변환 (argparse에서 str로 파싱됨)
arc_count = int(args.info_count) if args.info_count is not None else None
# .metadata 옵션이 활성화된 경우
if args.bMetadata:
ProcessMetadataToCBZ(archive_path, args.bPrint,
arc_volume, arc_number, args.info_count,
arc_volume, arc_number, arc_count,
arc_title, arc_writer, arc_series,
arc_tags if arc_tags else None)
return
@@ -114,8 +118,27 @@ def process_archive_file(archive_path: str, args: argparse.Namespace):
util.DbgOut(f"comicinfo.xml 발견됨: {archive_path} (경로: {comicinfo_path_in_zip})", True)
comicinfo_xml_bytes = util.GetZippedFileByte(archive_path, comicinfo_path_in_zip)
if comicinfo_xml_bytes:
fixed_xml = None
if args.bFixXml:
fixed_xml, msgs = ValidateAndFixComicInfoXML(comicinfo_xml_bytes)
for msg in msgs:
util.DbgOut(msg, True)
if fixed_xml is not None:
comicinfo_xml_bytes = fixed_xml
has_mods = any([arc_title, arc_writer, arc_series, arc_tags,
arc_number, arc_volume, arc_count])
if not has_mods:
if fixed_xml is not None:
util.AddFileToZip(archive_path, comicinfo_path_in_zip, fixed_xml)
util.DbgOut(f"ComicInfo.xml 특수문자 수정 완료: {archive_path}", True)
if args.bPrint:
print(comicinfo_xml_bytes.decode('utf-8'))
return
modified = ModComicInfoXML(comicinfo_xml_bytes, arc_title, arc_writer, arc_series, arc_tags,
info_number=arc_number, info_volume=arc_volume, info_count=args.info_count)
info_number=arc_number, info_volume=arc_volume, info_count=arc_count)
if args.bPrint:
output = modified.decode('utf-8') if modified else comicinfo_xml_bytes.decode('utf-8')
@@ -123,9 +146,12 @@ def process_archive_file(archive_path: str, args: argparse.Namespace):
elif modified:
util.AddFileToZip(archive_path, comicinfo_path_in_zip, modified)
util.DbgOut(f"ComicInfo.xml 업데이트됨: {archive_path} (경로: {comicinfo_path_in_zip})", True)
elif fixed_xml is not None:
util.AddFileToZip(archive_path, comicinfo_path_in_zip, fixed_xml)
util.DbgOut(f"ComicInfo.xml 특수문자 수정 완료: {archive_path}", True)
else:
util.DbgOut(f"comicinfo.xml 없음: {archive_path}", True)
if arc_volume or arc_number or arc_title or arc_writer or arc_series or arc_tags or args.info_count:
if arc_volume or arc_number or arc_title or arc_writer or arc_series or arc_tags or arc_count is not None:
new_xml = CreateComicInfoXML(
info_title=arc_title or "",
info_writer=arc_writer or "",
@@ -134,7 +160,7 @@ def process_archive_file(archive_path: str, args: argparse.Namespace):
pages=[],
volume=arc_volume or "",
number=arc_number or "",
count=int(args.info_count) if args.info_count is not None else -1
count=arc_count if arc_count is not None else -1
)
if new_xml:
if args.bPrint:
@@ -150,7 +176,9 @@ def ExpandPathVars(text: str | None, filepath: str) -> str | None:
"""
if not text:
return text
dirpath = os.path.dirname(filepath)
# 경로 정규화: ./ 또는 ../ 등 상대 경로 표현을 제거하여 정확한 폴더명 추출
normpath = os.path.normpath(filepath)
dirpath = os.path.dirname(normpath)
cur_folder = os.path.basename(dirpath) if dirpath else ""
par_folder = os.path.basename(os.path.dirname(dirpath)) if dirpath and os.path.dirname(dirpath) else ""
filename_no_ext = os.path.splitext(os.path.basename(filepath))[0]
@@ -170,8 +198,8 @@ def ExpandPathVars(text: str | None, filepath: str) -> str | None:
for key, value in var_map.items():
result = result.replace(f"<{key}>", value)
# After variable expansion, check for :s/pattern/replacement/ suffix
m = re.match(r'^(.*):s(.)(.+?)\2(.*?)(?:\2([ig]*))?$', result, re.DOTALL)
# After variable expansion, check for :s/pattern/replacement/ or s/pattern/replacement/ suffix
m = re.match(r'^(.*):?s(.)(.+?)\2(.*?)(?:\2([ig]*))?$', result, re.DOTALL)
if m:
source = m.group(1)
delim = m.group(2)
@@ -183,10 +211,12 @@ def ExpandPathVars(text: str | None, filepath: str) -> str | None:
if 'i' in flags_str:
flags |= re.IGNORECASE
count = 0 if 'g' in flags_str else 1
try:
result = re.sub(pattern, replacement_safe, source, count=count, flags=flags)
except re.error:
pass
# source가 비어있으면 정규식 미적용 (ResolveRegexTransform에서 처리)
if source:
try:
result = re.sub(pattern, replacement_safe, source, count=count, flags=flags)
except re.error:
pass
return result
@@ -255,6 +285,38 @@ def ResolveRegexTransform(text: str | None, current_value: str | None) -> str |
return text
def ValidateAndFixComicInfoXML(xml_bytes: bytes) -> tuple[bytes | None, list[str]]:
try:
ET.fromstring(xml_bytes)
return (None, ["XML is well-formed."])
except ET.ParseError as e:
msgs = [f"Parse error: {e}"]
try:
text = xml_bytes.decode('utf-8')
except UnicodeDecodeError:
msgs.append("Cannot decode XML (expected UTF-8).")
return (None, msgs)
text, count = re.subn(
r'&(?!(?:amp|lt|gt|quot|apos);|#\d+;|#x[0-9a-fA-F]+;)',
'&amp;',
text
)
if count > 0:
msgs.append(f"Fixed {count} unescaped & character(s).")
fixed_bytes = text.encode('utf-8')
try:
ET.fromstring(fixed_bytes)
msgs.append("XML is now well-formed.")
except ET.ParseError as e2:
msgs.append(f"Remaining issues after fix: {e2}")
return (fixed_bytes, msgs)
msgs.append("No fixable issues found.")
return (None, msgs)
def ParsePathForComicInfo(strFullPath: str) -> dict:
"""
파일 경로에서 작가, 제목, 번호를 추출합니다.
@@ -440,7 +502,7 @@ def ConvertLanguageToISO(lang_name: str) -> str:
#
def ProcessMetadataToCBZ(strArcPath: str, bPrint: bool = False,
volume: str = "-1", number: str = "", count: int = -1,
volume: str = "-1", number: str = "", count: int | None = None,
info_title: str = None, info_writer: str = None,
info_series: str = None, tags: list[str] = None) -> bool:
"""