From 8d3245b1979eb37cac1eb3967a6a1168d58424da Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EA=B9=80=EC=A4=80=ED=98=B8?= Date: Fri, 31 Jul 2026 07:04:50 +0000 Subject: [PATCH] Initial commit --- .gitignore | 75 + Docs/APPNOTE-6.39.9-KR.md | 2323 +++++++++++++++++++++++ Docs/APPNOTE-6.39.9.txt | 3778 +++++++++++++++++++++++++++++++++++++ LICENSE | 18 + MyZip.sln | 31 + MyZip.vcxproj | 131 ++ MyZip.vcxproj.filters | 22 + README.md | 3 + main.cpp | 6 + 9 files changed, 6387 insertions(+) create mode 100644 .gitignore create mode 100644 Docs/APPNOTE-6.39.9-KR.md create mode 100644 Docs/APPNOTE-6.39.9.txt create mode 100644 LICENSE create mode 100644 MyZip.sln create mode 100644 MyZip.vcxproj create mode 100644 MyZip.vcxproj.filters create mode 100644 README.md create mode 100644 main.cpp diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c9af22c --- /dev/null +++ b/.gitignore @@ -0,0 +1,75 @@ +# User-specific files +*.rsuser +*.suo +*.user +*.userosscache +*.sln.docstates + +# Build results +[Dd]ebug/ +[Dd]ebugPublic/ +[Rr]elease/ +[Rr]eleases/ +x64/ +x86/ +[Ww][Ii][Nn]32/ +[Aa][Rr][Mm]/ +[Aa][Rr][Mm]64/ +bld/ +[Bb]in/ +[Oo]bj/ +[Ll]og/ +[Ll]ogs/ + +# Visual Studio 2015/2017 cache/options directory +.vs/ + +# MSTest test Results +[Tt]est[Rr]esult*/ +[Bb]uild[Ll]og.* + +# Files built by Visual Studio +*_i.c +*_p.c +*_h.h +*.ilk +*.meta +*.obj +*.iobj +*.pch +*.pdb +*.ipdb +*.pgc +*.pgd +*.rsp +# but not Directory.Build.rsp, as it configures directory-level build defaults +!Directory.Build.rsp +*.sbr +*.tlb +*.tli +*.tlh +*.tmp +*.tmp_proj +*_wpftmp.csproj +*.log +*.tlog +*.vspscc +*.vssscc +.builds +*.pidb +*.svclog +*.scc + +# Visual C++ cache files +ipch/ +*.aps +*.ncb +*.opendb +*.opensdf +*.sdf +*.cachefile +*.VC.db +*.VC.VC.opendb + +# Zip Foramt Test +Test \ No newline at end of file diff --git a/Docs/APPNOTE-6.39.9-KR.md b/Docs/APPNOTE-6.39.9-KR.md new file mode 100644 index 0000000..5301128 --- /dev/null +++ b/Docs/APPNOTE-6.39.9-KR.md @@ -0,0 +1,2323 @@ +# .ZIP 파일 형식 명세서 (APPNOTE.TXT) + +- **버전**: 6.3.9 +- **상태**: FINAL - 6.3.8 버전을 대체함 +- **개정일**: 2020년 7월 15일 +- **저작권**: Copyright (c) 1989 - 2014, 2018, 2019, 2020 PKWARE Inc., All Rights Reserved. + +> 본 문서는 PKWARE의 원본 영문 명세서 "APPNOTE.TXT - .ZIP File Format Specification"을 한국어로 번역한 것입니다. + +--- + +## 1.0 서론 + +### 1.1 목적 + +**1.1.1** 본 명세서는 플랫폼 간 상호운용이 가능한 파일 저장 및 전송 형식을 정의하는 것을 목적으로 합니다. 1989년 최초 발표 이후 PKWARE, Inc.("PKWARE")는 본 명세서를 주기적으로 발행하고 유지관리함으로써 .ZIP 파일 형식의 상호운용성을 보장하는 데 전념해 왔습니다. 이 형식을 사용하여 이익을 얻는 모든 .ZIP 호환 벤더 및 애플리케이션 개발자가 이러한 상호운용성에 대한 노력을 공유하고 지지해 주기를 기대합니다. + +### 1.2 범위 + +**1.2.1** ZIP은 가장 널리 사용되는 압축 파일 형식 중 하나입니다. 여러 파일을 하나의 상호운용 가능한 컨테이너로 모으고, 압축하고, 암호화하는 데 보편적으로 사용됩니다. 이 형식은 특정 용도나 애플리케이션 요구사항을 정의하지 않으며, 특정 구현 지침도 제공하지 않습니다. 본 문서는 ZIP 파일을 생성하기 위한 저장 형식에 대한 세부 정보를 제공합니다. ZIP 파일이 무엇인지를 설명하는 레코드 및 필드에 대한 정보를 제공합니다. + +### 1.3 상표 + +**1.3.1** PKWARE, PKZIP, Smartcrypt, SecureZIP, PKSFX는 미국 및 기타 국가에서 PKWARE, Inc.의 등록 상표입니다. PKPatchMaker, Deflate64, ZIP64는 PKWARE, Inc.의 상표입니다. 본 문서에서 언급되는 그 밖의 상표는 식별 목적으로만 표기된 것이며 각 소유자의 재산입니다. + +### 1.4 허용된 사용 + +**1.4.1** 본 문서 "APPNOTE.TXT - .ZIP File Format Specification"은 PKWARE의 독점 재산입니다. 본 문서에 포함된 정보의 사용은 ZIP 형식으로 파일을 읽고 쓰는 제품, 프로그램 및 프로세스를 만드는 목적에 한해, 여기에 명시된 조건에 따라 허용됩니다. + +**1.4.2** 본 문서의 내용을 다른 간행물에서 사용하는 것은 본 문서를 참조로 인용하는 경우에만 허용됩니다. PKWARE의 사전 서면 허가 없이 본 문서의 전체 또는 일부를 복제하거나 배포하는 것은 엄격히 금지됩니다. + +**1.4.3** 본 문서에서 제공하는 일부 기술적 구성 요소는 PKWARE의 특허받은 독점 기술이며, 이를 사용하려면 PKWARE와 별도로 체결한 실행 라이선스 계약이 필요합니다. 해당하는 구성 요소에는 다음과 유사한 문구가 표기되어 있습니다: '본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오.' + +### 1.5 PKWARE 연락처 + +**1.5.1** 본 형식, 그 사용, 라이선스에 대해 질문이 있거나 결함을 보고하거나 변경/추가를 요청하고자 하는 경우 다음으로 연락하십시오: + +``` +PKWARE, Inc. +201 E. Pittsburgh Avenue, Suite 400 +Milwaukee, WI 53204 ++1-414-289-9788 ++1-414-289-9789 (FAX) +zipformat@pkware.com +``` + +**1.5.2** 이 형식에 대한 정보 및 본 문서의 참조 사본은 다음에서 공개적으로 확인할 수 있습니다: + +``` +http://www.pkware.com/appnote +``` + +### 1.6 면책 조항 + +**1.6.1** PKWARE는 파일 형식, 알고리즘 및 관련 프로그램에 관한 최신의 정확한 정보를 제공하고자 노력하지만, 오류나 누락의 가능성을 완전히 배제할 수는 없습니다. 따라서 PKWARE는 관련 프로그램, 해당 프로그램이 생성하거나 접근하는 파일의 형식, 해당 프로그램이 사용하는 알고리즘, 또는 그 밖의 어떠한 사항에 관해서도 그 정보가 최신이거나, 정확하거나, 올바르다는 어떠한 보증도 명시적으로 부인합니다. 부정확한 정보로 인한 손해의 위험은 정보를 사용하는 사용자가 전적으로 부담합니다. 또한 관련 프로그램 및/또는 해당 프로그램이 생성하거나 접근하는 파일 형식 및/또는 해당 프로그램이 사용하는 알고리즘에 관한 정보는 예고 없이 변경될 수 있습니다. + +--- + +## 2.0 개정 내역 + +### 2.1 문서 상태 + +**2.1.1** 본 파일의 상태(STATUS)가 DRAFT로 표시된 경우, 그 내용은 ZIP 형식 자체에 대한 변경이나 본 문서의 기타 내용 변경으로 구성되는 제안된 개정 사항을 정의합니다. DRAFT 상태의 문서 및 형식 버전은 FINAL 상태로 발행되기 전에 수정될 수 있습니다. DRAFT 버전은 ZIP 커뮤니티에 예정된 변경 사항을 통지하고 검토 및 의견 제시 기회를 제공하기 위해 주기적으로 발행됩니다. + +**2.1.2** 상태가 FINAL로 표시된 문서 버전은 해당 버전의 문서에 대해 최종 형식으로 간주되며, 더 높은 버전 번호를 가진 새 문서가 발행되기 전까지는 추가로 변경되지 않습니다. 이 형식 명세서의 최신 버전은 기술적으로 가능한 한 이전 모든 버전과 상호운용성을 유지하도록 의도되었습니다. + +### 2.2 변경 이력 + +| 버전 | 변경 내용 | 날짜 | +|---|---|---| +| 5.2 | 단일 비밀번호 대칭 암호화 저장 방식 도입 | 2003.07.16 | +| 6.1.0 | 스마트카드 호환성 추가, 인증서 저장에 관한 문서화 | 2004.01.20 | +| 6.2.0 | 메타데이터 암호화를 위한 중앙 디렉터리 암호화 도입, "Version Made By" 값에 OS X 추가 | 2004.04.26 | +| 6.2.1 | ID 0x4690을 사용하는 POSZIP용 Extra Field 자리 표시자 추가, "zip64 end of central directory record"의 크기 필드 명확화 | 2005.04.01 | +| 6.2.2 | 강력 암호화(Strong Encryption)의 최종 기능 명세 문서화, 명확화 및 오탈자 수정 | 2006.01.06 | +| 6.3.0 | 테이프 위치 지정 저장 파라미터 추가, 지원 해시 알고리즘 목록 확대, 지원 압축 알고리즘 목록 확대, 지원 암호화 알고리즘 목록 확대, 유니코드 파일명 저장 옵션 추가, 데이터 디스크립터 레코드의 일관된 사용에 대한 명확화, "Extra Field" 정의 추가 | 2006.09.29 | +| 6.3.1 | SHA-256/384/512의 표준 해시 값 수정 | 2007.04.11 | +| 6.3.2 | 압축 방법 97 추가, UTF-8 파일명 및 파일 코멘트 저장을 위한 InfoZIP "Extra Field" 값 문서화 | 2007.09.28 | +| 6.3.3 | 다른 문서 및 표준에서 본 APPNOTE를 더 쉽게 참조할 수 있도록 서식 변경 | 2012.09.01 | +| 6.3.4 | 주소 변경 | 2014.10.01 | +| 6.3.5 | 압축 방법 16, 99 문서화(4.4.5, 4.6.1, 5.11, 5.17, APPENDIX E), 여러 오탈자 수정(2.1.2, 3.2, 4.1.1, 10.2), 레거시 알고리즘을 더 이상 사용에 적합하지 않은 것으로 표기(4.4.5.1), MS-DOS 시간 형식에 대한 명확화 추가(4.4.6), 타임스탬프용 Extra Field ID 할당(4.5.2), 필드 코드 설명 수정(A.2), MAY/SHOULD/MUST의 일관된 사용, 0x0065 레코드 속성 코드 확대(B.2), 0x0022 Extra Data에 대한 초기 정보 추가 | 2018.11.31 | +| 6.3.6 | 오탈자 수정(4.4.1.3) | 2019.04.26 | +| 6.3.7 | Zstandard 압축 방법 ID 추가(4.4.5), 여러 오탈자 수정, 범용 비트 플래그 14의 용도 명시, Data Stream Alignment Extra Data 정보 추가(4.6.11) | (미기재) | +| 6.3.8 | Zstandard 압축 방법 ID 충돌 해결(4.4.5), 사용 중인 추가 압축 방법 ID 값 추가 | (미기재) | +| 6.3.9 | Data Stream Alignment 설명의 오탈자 수정(4.6.11) | (미기재) | + +--- + +## 3.0 표기법 + +**3.1** MUST 또는 SHALL 용어는 필수 요소를 나타냅니다. + +**3.2** MUST NOT 또는 SHALL NOT은 사용이 금지된 요소를 나타냅니다. + +**3.3** SHOULD는 권장(RECOMMENDED) 요소를 나타냅니다. + +**3.4** SHOULD NOT은 사용이 권장되지 않는(NOT RECOMMENDED) 요소를 나타냅니다. + +**3.5** MAY는 선택적(OPTIONAL) 요소를 나타냅니다. + +--- + +## 4.0 ZIP 파일 + +### 4.1 ZIP 파일이란 + +**4.1.1** ZIP 파일은 표준 .ZIP 파일 확장자로 식별될 수 있지만, 확장자 사용이 필수는 아닙니다. .ZIPX 확장자 사용도 ZIP 파일로 인정되며 사용할 수 있습니다. ZIP 형식을 사용하는 그 밖의 일반적인 파일 확장자로는 .JAR, .WAR, .DOCX, .XLSX, .PPTX, .ODT, .ODS, .ODP 등이 있습니다. ZIP 파일을 읽거나 쓰는 프로그램은 이 형식의 파일을 식별하기 위해 본 문서에 설명된 내부 레코드 시그니처에 의존해야 합니다(SHOULD). + +**4.1.2** ZIP 파일은 최소 하나의 파일을 포함해야 하며(SHOULD), 여러 파일을 포함할 수 있습니다(MAY). + +**4.1.3** ZIP 파일에 넣는 파일의 크기를 줄이기 위해 데이터 압축을 사용할 수 있으나(MAY), 필수는 아닙니다. 이 형식은 여러 데이터 압축 알고리즘의 사용을 지원합니다. 압축을 사용하는 경우 문서화된 압축 알고리즘 중 하나를 사용해야 합니다(MUST). 구현자는 자신의 데이터로 실험하여 자신의 요구에 가장 적합한 압축률을 제공하는 알고리즘을 파악할 것을 권장합니다. 압축 방법 8(Deflate)은 대부분의 ZIP 호환 애플리케이션 프로그램이 기본으로 사용하는 방법입니다. + +**4.1.4** ZIP 파일 내 파일을 보호하기 위해 데이터 암호화를 사용할 수 있습니다(MAY). 이 형식에서 지원하는 암호화 키 방식으로는 비밀번호와 공개/개인 키가 있습니다. 둘 중 하나를 개별적으로 사용하거나 조합하여 사용할 수 있습니다(MAY). 암호화는 개별 파일에 적용할 수 있습니다(MAY). 중앙 디렉터리(Central Directory)에 저장된 ZIP 파일 메타데이터를 암호화하여 추가 보안을 적용할 수 있습니다(MAY). 자세한 내용은 강력 암호화 명세(Strong Encryption Specification) 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.1.5** 각 파일에 대해 CRC32를 사용한 데이터 무결성이 반드시 제공되어야 합니다(MUST). + +**4.1.6** 디지털 서명을 사용하여 추가적인 데이터 무결성을 포함할 수 있습니다(MAY). 개별 파일은 하나 이상의 디지털 서명으로 서명될 수 있습니다(MAY). 중앙 디렉터리는 서명하는 경우 단일 서명을 사용해야 합니다(MUST). + +**4.1.7** 파일은 압축되지 않은 상태(저장, stored)로 ZIP 파일 안에 배치될 수 있습니다(MAY). 본 문서에서 "저장(stored)"이라는 용어는 파일이 압축되지 않은 채로 ZIP 파일에 복사되는 것을 의미합니다. + +**4.1.8** ZIP 파일에 배치되는 각 데이터 파일은 동일한 ZIP 파일 내 다른 데이터 파일이 어떻게 저장되는지와 무관하게, 독립적으로 압축, 저장, 암호화 또는 디지털 서명될 수 있습니다(MAY). + +**4.1.9** ZIP 파일은 스트리밍되거나, (고정 또는 이동식 매체 상에서) 세그먼트로 분할되거나, "자기 추출형(self-extracting)"으로 만들어질 수 있습니다(MAY). 자기 추출형 ZIP 파일은 대상 플랫폼용 추출 코드를 ZIP 파일 안에 반드시 포함해야 합니다(MUST). + +**4.1.10** 플랫폼 또는 애플리케이션별 요구를 위해 사용자 지정 목적으로 정의할 수 있는 extra data 필드를 통해 확장성이 제공됩니다(MAY). extra data 정의는 기존에 문서화된 레코드 정의와 충돌해서는 안 됩니다(MUST NOT). + +**4.1.11** ZIP의 일반적인 용도에는 매니페스트(manifest) 파일의 사용도 포함될 수 있습니다(MAY). 매니페스트 파일은 ZIP 파일 안에 저장된 파일 내에 애플리케이션별 정보를 저장합니다. 이 매니페스트 파일은 ZIP 파일의 첫 번째 파일이어야 합니다(SHOULD). 본 명세서는 ZIP 파일 내 매니페스트 파일 사용에 대한 정보나 지침을 제공하지 않습니다. 매니페스트 파일 사용에 대한 정보 및 애플리케이션 내 ZIP 사용에 대한 추가적인 프로필 정보는 애플리케이션 개발자에게 문의하십시오. + +**4.1.12** ZIP 파일은 다른 ZIP 파일 안에 넣을 수 있습니다(MAY). + +### 4.2 ZIP 메타데이터 + +**4.2.1** ZIP 파일은 ZIP 파일에 넣어진 파일들을 관리하는 데 필요한 저장 정보를 담은, 정의된 레코드 유형으로 구성된 메타데이터로 식별됩니다. 각 레코드 유형은 해당 레코드 유형을 식별하는 헤더 시그니처를 통해 반드시 식별되어야 합니다(MUST). 시그니처 값은 문자 "PK"를 나타내는 2바이트 상수 마커 0x4b50으로 시작합니다. + +### 4.3 .ZIP 파일의 일반 형식 + +**4.3.1** ZIP 파일은 "end of central directory record"를 반드시 포함해야 합니다(MUST). "end of central directory record"만 포함하는 ZIP 파일은 빈 ZIP 파일로 간주됩니다. 파일은 ZIP 파일 내에서 추가, 교체 또는 삭제될 수 있습니다(MAY). ZIP 파일은 오직 하나의 "end of central directory record"만 가져야 합니다(MUST). 본 명세서에 정의된 다른 레코드들은 개별 ZIP 파일의 저장 요구에 따라 필요한 만큼 사용될 수 있습니다(MAY). + +**4.3.2** ZIP 파일에 배치되는 각 파일 앞에는 해당 파일에 대한 "local file header" 레코드가 반드시 와야 합니다(MUST). 각 "local file header"는 ZIP 파일의 중앙 디렉터리 섹션 내 해당하는 "central directory header" 레코드를 반드시 동반해야 합니다(MUST). + +**4.3.3** 파일은 ZIP 파일 내에서 임의의 순서로 저장될 수 있습니다(MAY). ZIP 파일은 여러 볼륨에 걸쳐 있을 수 있으며(MAY), 사용자 정의 세그먼트 크기로 분할될 수도 있습니다(MAY). 본 문서에서 특정 데이터 요소에 대해 별도로 명시하지 않는 한, 모든 값은 리틀 엔디안(little-endian) 바이트 순서로 저장되어야 합니다(MUST). + +**4.3.4** "local file header", "encryption header", "end of central directory record"에는 압축을 적용해서는 안 됩니다(MUST NOT). 개별 "central directory record"는 압축되어서는 안 되지만(MUST NOT), 모든 central directory record의 집합은 압축될 수 있습니다(MAY). + +**4.3.5** 파일 데이터 뒤에는 해당 파일에 대한 "data descriptor"가 이어질 수 있습니다(MAY). 데이터 디스크립터는 ZIP 파일 스트리밍을 용이하게 하기 위해 사용됩니다. + +**4.3.6 전체 .ZIP 파일 형식:** + +``` +[local file header 1] +[encryption header 1] +[file data 1] +[data descriptor 1] +. +. +. +[local file header n] +[encryption header n] +[file data n] +[data descriptor n] +[archive decryption header] +[archive extra data record] +[central directory header 1] +. +. +. +[central directory header n] +[zip64 end of central directory record] +[zip64 end of central directory locator] +[end of central directory record] +``` + +**4.3.7 Local file header:** + +``` +local file header signature 4 bytes (0x04034b50) +version needed to extract 2 bytes +general purpose bit flag 2 bytes +compression method 2 bytes +last mod file time 2 bytes +last mod file date 2 bytes +crc-32 4 bytes +compressed size 4 bytes +uncompressed size 4 bytes +file name length 2 bytes +extra field length 2 bytes + +file name (variable size) +extra field (variable size) +``` + +**4.3.8 파일 데이터** + +파일에 대한 로컬 헤더 바로 뒤에는 해당 파일의 압축된 또는 저장된 데이터가 와야 합니다(SHOULD). 파일이 암호화된 경우, 파일에 대한 암호화 헤더는 로컬 헤더 뒤, 파일 데이터 앞에 위치해야 합니다(SHOULD). [local file header][encryption header][file data][data descriptor]의 연속은 .ZIP 아카이브 내 각 파일마다 반복됩니다. + +크기가 0바이트인 파일, 디렉터리, 그리고 그 밖에 내용이 없는 파일 유형은 파일 데이터를 포함해서는 안 됩니다(MUST NOT). + +**4.3.9 Data descriptor:** + +``` +crc-32 4 bytes +compressed size 4 bytes +uncompressed size 4 bytes +``` + +**4.3.9.1** 이 디스크립터는 범용 비트 플래그의 비트 3이 설정된 경우 반드시 존재해야 합니다(MUST, 아래 참고). 이는 바이트 정렬되어 있으며 압축 데이터의 마지막 바이트 바로 뒤에 옵니다. 이 디스크립터는 출력 .ZIP 파일에서 탐색(seek)이 불가능했던 경우, 예를 들어 출력 .ZIP 파일이 표준 출력이거나 탐색 불가능한 장치인 경우에만 사용해야 합니다(SHOULD). ZIP64(tm) 형식 아카이브의 경우, 압축 및 압축 해제 크기는 각각 8바이트입니다. + +**4.3.9.2** 파일을 압축할 때, 파일 크기가 0xFFFFFFFF를 초과하는 경우 압축/비압축 크기는 ZIP64 형식(8바이트 값)으로 저장해야 합니다(SHOULD). 다만 파일 크기와 무관하게 ZIP64 형식을 사용할 수도 있습니다(MAY). 추출 시 해당 파일에 대해 zip64 extended information extra field가 존재하면 압축/비압축 크기는 8바이트 값이 됩니다. + +**4.3.9.3** 원래는 시그니처가 할당되지 않았지만, 값 0x08074b50이 데이터 디스크립터 레코드의 시그니처 값으로 일반적으로 채택되어 사용되고 있습니다. 구현자는 ZIP 파일에 이 시그니처가 있거나 없는 상태로 데이터 디스크립터가 표시될 수 있음을 인지해야 하며(SHOULD), 호환성을 보장하기 위해 ZIP 파일을 읽을 때 두 경우 모두를 고려해야 합니다(SHOULD). + +**4.3.9.4** ZIP 파일을 작성할 때 구현자는 데이터 디스크립터 레코드를 표시하는 시그니처 값을 포함해야 합니다(SHOULD). 시그니처를 사용하는 경우, 데이터 디스크립터 레코드에 현재 정의된 필드들이 시그니처 바로 뒤에 옵니다. + +**4.3.9.5** 확장 가능한 데이터 디스크립터(extensible data descriptor)는 본 APPNOTE의 향후 버전에서 발표될 예정입니다. 이 새 레코드는 향후 이 레코드 사용과 관련된 충돌을 해결하고, 스트리밍 파일 처리에 대한 더 나은 지원을 제공하기 위한 것입니다. + +**4.3.9.6** 중앙 디렉터리 암호화 방식을 사용하는 경우, 데이터 디스크립터 레코드는 필수는 아니지만 사용할 수 있습니다(MAY). 이 레코드가 존재하고 범용 비트 플래그의 비트 3이 그 존재를 나타내도록 설정된 경우, 데이터 디스크립터 레코드의 필드 값들은 반드시 이진수 0으로 설정되어야 합니다(MUST). 자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.3.10 Archive decryption header:** + +**4.3.10.1** Archive Decryption Header는 ZIP 형식 명세서 버전 6.2에서 도입되었습니다. 이 레코드는 본 문서에서 설명하는 강력 암호화 명세(Strong Encryption Specification)의 일부로 구현된 중앙 디렉터리 암호화 기능을 지원하기 위해 존재합니다. 중앙 디렉터리 구조가 암호화된 경우, 이 decryption header는 암호화된 데이터 세그먼트 앞에 반드시 와야 합니다(MUST). + +**4.3.10.2** 암호화된 데이터 세그먼트는 (존재하는 경우) Archive extra data record와 암호화된 중앙 디렉터리 구조 데이터로 구성됩니다(SHALL). 이 데이터 레코드의 형식은 압축된 파일 데이터 앞에 오는 Decryption header 레코드와 동일합니다. 중앙 디렉터리 구조가 암호화된 경우, 이 데이터 레코드가 시작되는 위치는 Zip64 End of Central Directory record의 Start of Central Directory 필드를 사용하여 결정됩니다. 강력 암호화 명세에서 사용하는 Archive Decryption Header 레코드의 필드에 대한 정보는 해당 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.3.11 Archive extra data record:** + +``` +archive extra data signature 4 bytes (0x08064b50) +extra field length 4 bytes +extra field data (variable size) +``` + +**4.3.11.1** Archive Extra Data Record는 ZIP 형식 명세서 버전 6.2에서 도입되었습니다. 이 레코드는 본 문서에서 설명하는 강력 암호화 명세의 일부로 구현된 중앙 디렉터리 암호화 기능을 지원하기 위해 사용될 수 있습니다(MAY). 존재하는 경우, 이 레코드는 중앙 디렉터리 데이터 구조 바로 앞에 반드시 위치해야 합니다(MUST). + +**4.3.11.2** 이 데이터 레코드의 크기는 End of Central Directory record의 Size of the Central Directory 필드에 포함되어야 합니다(SHALL). 중앙 디렉터리 구조가 압축되었지만 암호화되지는 않은 경우, 이 데이터 레코드가 시작되는 위치는 Zip64 End of Central Directory record의 Start of Central Directory 필드를 사용하여 결정됩니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.3.12 Central directory structure:** + +``` +[central directory header 1] +. +. +. +[central directory header n] +[digital signature] +``` + +**File header:** + +``` +central file header signature 4 bytes (0x02014b50) +version made by 2 bytes +version needed to extract 2 bytes +general purpose bit flag 2 bytes +compression method 2 bytes +last mod file time 2 bytes +last mod file date 2 bytes +crc-32 4 bytes +compressed size 4 bytes +uncompressed size 4 bytes +file name length 2 bytes +extra field length 2 bytes +file comment length 2 bytes +disk number start 2 bytes +internal file attributes 2 bytes +external file attributes 4 bytes +relative offset of local header 4 bytes + +file name (variable size) +extra field (variable size) +file comment (variable size) +``` + +**4.3.13 Digital signature:** + +``` +header signature 4 bytes (0x05054b50) +size of data 2 bytes +signature data (variable size) +``` + +버전 6.2에서 중앙 디렉터리 암호화(Central Directory Encryption) 기능이 도입됨에 따라, 중앙 디렉터리 구조는 압축과 암호화가 모두 적용된 상태로 저장될 수 있습니다(MAY). 필수는 아니지만, 중앙 디렉터리 구조를 암호화할 때는 저장 효율을 높이기 위해 압축도 함께 적용된다고 가정합니다. 중앙 디렉터리 암호화 기능에 대한 정보는 강력 암호화 명세를 설명하는 섹션에서 확인할 수 있습니다. Digital Signature 레코드는 압축되지도 암호화되지도 않습니다. + +**4.3.14 Zip64 end of central directory record** + +``` +zip64 end of central dir +signature 4 bytes (0x06064b50) +size of zip64 end of central +directory record 8 bytes +version made by 2 bytes +version needed to extract 2 bytes +number of this disk 4 bytes +number of the disk with the +start of the central directory 4 bytes +total number of entries in the +central directory on this disk 8 bytes +total number of entries in the +central directory 8 bytes +size of the central directory 8 bytes +offset of start of central +directory with respect to +the starting disk number 8 bytes +zip64 extensible data sector (variable size) +``` + +**4.3.14.1** "size of zip64 end of central directory record"에 저장되는 값은 나머지 레코드의 크기여야 하며(SHOULD), 선행하는 12바이트는 포함하지 않아야 합니다(SHOULD NOT). + +``` +Size = SizeOfFixedFields + SizeOfVariableData - 12 +``` + +**4.3.14.2** 위 레코드 구조는 zip64 end of central directory record의 버전 1을 정의합니다. 버전 1은 ZIP64 대용량 파일 기능을 지원하기 위해 6.2 이전 버전의 본 명세서에 구현되었습니다. 버전 6.2에서 강력 암호화 명세의 일부로 구현된 중앙 디렉터리 암호화 기능은 이 레코드 구조의 버전 2를 정의합니다. 이 레코드의 버전 2 형식에 대한 세부 사항은 강력 암호화 명세를 설명하는 섹션을 참고하십시오. 버전 2 사용에 관한 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.3.14.3** 이 레코드의 버전 1 또는 버전 2 뒤에 오는 zip64 extensible data sector 필드에는 특수 목적 데이터가 존재할 수 있습니다(MAY). 이 특수 목적 데이터를 식별하기 위해서는 다음으로 구성된 식별 헤더 블록을 반드시 포함해야 합니다(MUST): + +``` +Header ID - 2 bytes +Data Size - 4 bytes +``` + +Header ID 필드는 뒤따르는 데이터 블록의 유형을 나타냅니다. + +Data Size는 해당 데이터 블록 유형에 대해 뒤따르는 바이트 수를 나타냅니다. + +**4.3.14.4** 여러 개의 특수 목적 데이터 블록이 존재할 수 있습니다(MAY). 각각은 Header ID와 Data Size 필드가 반드시 선행해야 합니다(MUST). 이 필드에서 지원되는 Header ID 값의 현재 매핑은 APPENDIX C에 정의되어 있습니다. + +**4.3.15 Zip64 end of central directory locator** + +``` +zip64 end of central dir locator +signature 4 bytes (0x07064b50) +number of the disk with the +start of the zip64 end of +central directory 4 bytes +relative offset of the zip64 +end of central directory record 8 bytes +total number of disks 4 bytes +``` + +**4.3.16 End of central directory record:** + +``` +end of central dir signature 4 bytes (0x06054b50) +number of this disk 2 bytes +number of the disk with the +start of the central directory 2 bytes +total number of entries in the +central directory on this disk 2 bytes +total number of entries in +the central directory 2 bytes +size of the central directory 4 bytes +offset of start of central +directory with respect to +the starting disk number 4 bytes +.ZIP file comment length 2 bytes +.ZIP file comment (variable size) +``` + +### 4.4 필드 설명 + +#### 4.4.1 필드에 대한 일반 참고사항 + +**4.4.1.1** 별도로 언급하지 않는 한 모든 필드는 부호 없는 값이며, 인텔의 low-byte:high-byte, low-word:high-word 순서로 저장됩니다. + +**4.4.1.2** 문자열 필드는 길이가 명시적으로 주어지므로 널(null) 문자로 종료되지 않습니다. + +**4.4.1.3** 중앙 디렉터리의 항목들은 .ZIP 파일 안에 실제 파일들이 나타나는 순서와 반드시 같지는 않을 수 있습니다. + +**4.4.1.4** end of central directory record의 필드 중 하나가 필요한 데이터를 담기에 너무 작은 경우, 해당 필드는 -1(0xFFFF 또는 0xFFFFFFFF)로 설정해야 하며(SHOULD), ZIP64 형식 레코드를 생성해야 합니다(SHOULD). + +**4.4.1.5** 아카이브를 분할하거나 스패닝할 때 end of central directory record와 Zip64 end of central directory locator record는 반드시 같은 디스크에 있어야 합니다(MUST). + +#### 4.4.2 version made by (2바이트) + +**4.4.2.1** 상위 바이트는 파일 속성 정보의 호환성을 나타냅니다. 외부 파일 속성이 MS-DOS와 호환되어 PKZIP for DOS 버전 2.04g로 읽을 수 있는 경우 이 값은 0입니다. 이러한 속성이 호환되지 않는 경우, 이 값은 해당 속성이 호환되는 호스트 시스템을 식별합니다. 소프트웨어는 이 정보를 사용하여 텍스트 파일 등의 라인 레코드 형식을 판단할 수 있습니다. + +**4.4.2.2** 현재 매핑은 다음과 같습니다: + +``` + 0 - MS-DOS 및 OS/2 (FAT / VFAT / FAT32 파일 시스템) + 1 - Amiga 2 - OpenVMS + 3 - UNIX 4 - VM/CMS + 5 - Atari ST 6 - OS/2 H.P.F.S. + 7 - Macintosh 8 - Z-System + 9 - CP/M 10 - Windows NTFS +11 - MVS (OS/390 - Z/OS) 12 - VSE +13 - Acorn Risc 14 - VFAT +15 - alternate MVS 16 - BeOS +17 - Tandem 18 - OS/400 +19 - OS X (Darwin) 20~255 - 사용 안 함 +``` + +**4.4.2.3** 하위 바이트는 파일을 인코딩하는 데 사용된 소프트웨어가 지원하는 ZIP 명세 버전(본 문서의 버전)을 나타냅니다. 값/10은 주(major) 버전 번호를, 값 mod 10은 부(minor) 버전 번호를 나타냅니다. + +#### 4.4.3 version needed to extract (2바이트) + +**4.4.3.1** 파일을 추출하는 데 필요한 최소 지원 ZIP 명세 버전이며, 위와 같이 매핑됩니다. 이 값은 ZIP 프로그램이 해당 파일을 추출하기 위해 반드시 지원해야 하는 특정 형식 기능을 기반으로 합니다. 파일에 여러 기능이 적용된 경우, 최소 버전은 가장 높은 값을 가진 기능으로 설정해야 합니다(MUST). 발행된 형식 명세에 영향을 미치는 새로운 기능이나 기능 변경은 충돌을 피하기 위해 마지막으로 발행된 값보다 높은 버전 번호를 사용하여 구현됩니다. + +**4.4.3.2** 현재 최소 기능 버전은 다음과 같이 정의됩니다: + +``` +1.0 - 기본값 +1.1 - 파일이 볼륨 레이블임 +2.0 - 파일이 폴더(디렉터리)임 +2.0 - 파일이 Deflate 압축을 사용하여 압축됨 +2.0 - 파일이 전통적인 PKWARE 암호화를 사용하여 암호화됨 +2.1 - 파일이 Deflate64(tm)를 사용하여 압축됨 +2.5 - 파일이 PKWARE DCL Implode를 사용하여 압축됨 +2.7 - 파일이 패치 데이터 세트임 +4.5 - 파일이 ZIP64 형식 확장을 사용함 +4.6 - 파일이 BZIP2 압축을 사용하여 압축됨* +5.0 - 파일이 DES를 사용하여 암호화됨 +5.0 - 파일이 3DES를 사용하여 암호화됨 +5.0 - 파일이 원본 RC2 암호화를 사용하여 암호화됨 +5.0 - 파일이 RC4 암호화를 사용하여 암호화됨 +5.1 - 파일이 AES 암호화를 사용하여 암호화됨 +5.1 - 파일이 수정된 RC2 암호화를 사용하여 암호화됨** +5.2 - 파일이 수정된 RC2-64 암호화를 사용하여 암호화됨** +6.1 - 파일이 non-OAEP 키 래핑을 사용하여 암호화됨*** +6.2 - 중앙 디렉터리 암호화 +6.3 - 파일이 LZMA를 사용하여 압축됨 +6.3 - 파일이 PPMd를 사용하여 압축됨+ +6.3 - 파일이 Blowfish를 사용하여 암호화됨 +6.3 - 파일이 Twofish를 사용하여 암호화됨 +``` + +**4.4.3.3 version needed to extract에 대한 참고사항** + +\* 초기 7.x(7.2 이전) 버전의 PKZIP은 BZIP2 압축에 대한 version needed to extract를 46이어야 함에도(SHOULD) 잘못되게 50으로 설정했습니다. + +\*\* RC2 수정 사항에 대한 추가 정보는 강력 암호화 명세 섹션을 참고하십시오. + +\*\*\* non-OAEP 키 래핑을 사용하는 인증서 암호화는 6.1 이상 모든 버전에서 의도된 동작 방식입니다. OAEP 키 래핑 지원은 6.1보다 오래된 버전의 PKZIP(5.0 또는 6.0)이 열게 될 ZIP 파일을 보낼 때의 하위 호환성 목적으로만 사용해야 합니다(MUST). + +\+ PPMd를 사용하여 압축된 파일은 version needed to extract 필드를 6.3으로 반드시 설정해야 하지만(MUST), 모든 ZIP 프로그램이 이를 강제하지는 않으며 이 값이 설정된 경우 PPMd로 압축된 데이터 파일의 압축 해제를 지원하지 못할 수 있습니다(MAY). + +ZIP64 확장을 사용하는 경우, zip64 end of central directory record의 해당 값도 반드시 함께 설정해야 합니다(MUST). 이 필드는 버전 1 형식과 버전 2 형식 중 어느 것이 사용되는지를 적절히 나타내도록 설정해야 합니다(SHOULD). + +#### 4.4.4 general purpose bit flag (2바이트) + +- **비트 0**: 설정된 경우 파일이 암호화되었음을 나타냅니다. + +- (방법 6 - Imploding용) **비트 1**: 압축 방법이 유형 6(Imploding)인 경우, 이 비트가 설정되어 있으면 8K 슬라이딩 딕셔너리가 사용되었음을, 설정되지 않았으면 4K 슬라이딩 딕셔너리가 사용되었음을 나타냅니다. + +- **비트 2**: 압축 방법이 유형 6(Imploding)인 경우, 이 비트가 설정되어 있으면 슬라이딩 딕셔너리 출력을 인코딩하는 데 3개의 Shannon-Fano 트리가 사용되었음을, 설정되지 않았으면 2개의 트리가 사용되었음을 나타냅니다. + +- (방법 8, 9 - Deflating용) + +| 비트 2 | 비트 1 | 의미 | +|---|---|---| +| 0 | 0 | 표준(Normal, -en) 압축 옵션 사용 | +| 0 | 1 | 최대(Maximum, -exx/-ex) 압축 옵션 사용 | +| 1 | 0 | 빠른(Fast, -ef) 압축 옵션 사용 | +| 1 | 1 | 초고속(Super Fast, -es) 압축 옵션 사용 | + +- (방법 14 - LZMA용) **비트 1**: 압축 방법이 유형 14(LZMA)인 경우, 이 비트가 설정되어 있으면 압축 데이터 스트림의 끝을 표시하는 EOS(end-of-stream) 마커가 사용됨을 나타냅니다. 설정되지 않았으면 EOS 마커가 없으며 추출을 위해서는 압축 데이터 크기를 알고 있어야 합니다. + + 참고: 비트 1과 2는 압축 방법이 그 밖의 다른 것인 경우 정의되지 않습니다. + +- **비트 3**: 이 비트가 설정된 경우 로컬 헤더의 crc-32, 압축 크기, 비압축 크기 필드는 모두 0으로 설정됩니다. 올바른 값은 압축 데이터 바로 뒤의 데이터 디스크립터에 기록됩니다. (참고: DOS용 PKZIP 버전 2.04g는 이 비트를 방법 8 압축에 대해서만 인식하며, 이후 버전의 PKZIP은 모든 압축 방법에 대해 이 비트를 인식합니다.) + +- **비트 4**: 방법 8을 위한 향상된 deflating(enhanced deflating) 용도로 예약되어 있습니다. + +- **비트 5**: 이 비트가 설정된 경우 파일이 압축된 패치 데이터임을 나타냅니다. (참고: PKZIP 버전 2.70 이상 필요) + +- **비트 6**: 강력 암호화(Strong encryption). 이 비트가 설정된 경우 version needed to extract 값을 최소 50으로 설정해야 하며(MUST), 비트 0도 함께 설정해야 합니다(MUST). AES 암호화를 사용하는 경우 version needed to extract 값은 최소 51이어야 합니다(MUST). 자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +- **비트 7**: 현재 사용되지 않음. +- **비트 8**: 현재 사용되지 않음. +- **비트 9**: 현재 사용되지 않음. +- **비트 10**: 현재 사용되지 않음. + +- **비트 11**: 언어 인코딩 플래그(EFS, Language encoding flag). 이 비트가 설정된 경우 해당 파일의 파일명과 코멘트 필드는 반드시 UTF-8로 인코딩되어야 합니다(MUST, APPENDIX D 참고). + +- **비트 12**: 향상된 압축을 위해 PKWARE가 예약함. + +- **비트 13**: 중앙 디렉터리를 암호화할 때 설정되며, 로컬 헤더의 선택된 데이터 값들이 실제 값을 숨기기 위해 마스킹되었음을 나타냅니다. 자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +- **비트 14**: 대체 스트림(alternate streams)을 위해 PKWARE가 예약함. + +- **비트 15**: PKWARE가 예약함. + +#### 4.4.5 compression method (2바이트) + +``` + 0 - 파일이 저장됨 (압축 없음) + 1 - 파일이 Shrunk 방식으로 압축됨 + 2 - 파일이 압축 계수 1의 Reduced 방식으로 압축됨 + 3 - 파일이 압축 계수 2의 Reduced 방식으로 압축됨 + 4 - 파일이 압축 계수 3의 Reduced 방식으로 압축됨 + 5 - 파일이 압축 계수 4의 Reduced 방식으로 압축됨 + 6 - 파일이 Imploded 방식으로 압축됨 + 7 - Tokenizing 압축 알고리즘용으로 예약됨 + 8 - 파일이 Deflated 방식으로 압축됨 + 9 - Deflate64(tm)를 사용하는 Enhanced Deflating +10 - PKWARE Data Compression Library Imploding (구 IBM TERSE) +11 - PKWARE 예약 +12 - 파일이 BZIP2 알고리즘을 사용하여 압축됨 +13 - PKWARE 예약 +14 - LZMA +15 - PKWARE 예약 +16 - IBM z/OS CMPSC 압축 +17 - PKWARE 예약 +18 - 파일이 IBM TERSE(신형)를 사용하여 압축됨 +19 - IBM LZ77 z Architecture +20 - 지원 중단(zstd는 방법 93 사용) +93 - Zstandard(zstd) 압축 +94 - MP3 압축 +95 - XZ 압축 +96 - JPEG variant +97 - WavPack 압축 데이터 +98 - PPMd version I, Rev 1 +99 - AE-x 암호화 마커 (APPENDIX E 참고) +``` + +**4.4.5.1** 방법 1~6은 레거시 알고리즘이며 파일 압축 시 더 이상 사용을 권장하지 않습니다. + +#### 4.4.6 date and time 필드 (각 2바이트) + +날짜와 시간은 표준 MS-DOS 형식으로 인코딩됩니다. 표준 입력(standard input)에서 입력이 온 경우, 날짜와 시간은 해당 데이터에 대한 압축이 시작된 시점의 값입니다. 중앙 디렉터리를 암호화하고 범용 비트 플래그 13이 설정되어 마스킹을 나타내는 경우, 로컬 헤더에 저장되는 값은 0이 됩니다. MS-DOS 시간 형식은 UTC와 같이 더 흔히 사용되는 컴퓨터 시간 형식과 다릅니다. 예를 들어 MS-DOS는 1980년을 기준으로 한 연도 값과 2초 단위의 정밀도를 사용합니다. + +#### 4.4.7 CRC-32 (4바이트) + +CRC-32 알고리즘은 David Schwaderer가 저술한 "C Programmers Guide to NetBIOS"(Howard W. Sams & Co. Inc. 발행)에서 관대하게 제공된 것입니다. CRC의 '매직 넘버'는 0xdebb20e3입니다. 적절한 CRC 사전/사후 조정이 사용되는데, 이는 CRC 레지스터가 모두 1(시작값 0xffffffff)로 사전 조정되고, 값은 CRC 잔여값의 1의 보수를 취하여 사후 조정됨을 의미합니다. 범용 플래그의 비트 3이 설정된 경우, 이 필드는 로컬 헤더에서 0으로 설정되며 올바른 값은 데이터 디스크립터와 중앙 디렉터리에 기록됩니다. 중앙 디렉터리를 암호화할 때, 로컬 헤더가 ZIP64 형식이 아니고 범용 비트 플래그 13이 설정되어 마스킹을 나타내는 경우 로컬 헤더에 저장되는 값은 0이 됩니다. + +#### 4.4.8 compressed size (4바이트) / 4.4.9 uncompressed size (4바이트) + +각각 압축된 파일의 크기(4.4.8)와 비압축 크기(4.4.9)입니다. decryption header가 존재하는 경우 파일 데이터 앞에 위치하며, 압축 파일 크기 값에는 decryption header의 바이트 수가 포함됩니다. 범용 비트 플래그의 비트 3이 설정된 경우, 이 필드들은 로컬 헤더에서 0으로 설정되며 올바른 값은 데이터 디스크립터와 중앙 디렉터리에 기록됩니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFFFFFF인 경우, 크기는 해당하는 8바이트 ZIP64 extended information extra field에 담깁니다. 중앙 디렉터리를 암호화할 때, 로컬 헤더가 ZIP64 형식이 아니고 범용 비트 플래그 13이 설정되어 마스킹을 나타내는 경우 로컬 헤더에 저장되는 비압축 크기 값은 0이 됩니다. + +#### 4.4.10 file name length (2바이트) / 4.4.11 extra field length (2바이트) / 4.4.12 file comment length (2바이트) + +각각 파일명, extra field, 코멘트 필드의 길이입니다. 디렉터리 레코드와 이 세 필드를 합한 길이는 일반적으로 65,535바이트를 초과하지 않아야 합니다(SHOULD NOT). 표준 입력에서 입력이 온 경우 파일명 길이는 0으로 설정됩니다. + +#### 4.4.13 disk number start (2바이트) + +이 파일이 시작되는 디스크 번호입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFF인 경우, 크기는 해당하는 4바이트 zip64 extended information extra field에 담깁니다. + +#### 4.4.14 internal file attributes (2바이트) + +비트 1과 2는 PKWARE 전용으로 예약되어 있습니다. + +**4.4.14.1** 이 필드의 최하위 비트는, 설정된 경우 파일이 겉보기에 ASCII 또는 텍스트 파일임을 나타내며, 설정되지 않은 경우 파일이 겉보기에 바이너리 데이터를 포함함을 나타냅니다. 나머지 비트는 버전 1.0에서 사용되지 않습니다. + +**4.4.14.2** 이 필드의 0x0002 비트는, 설정된 경우 각 논리 레코드 앞에 해당 레코드의 길이를 나타내는 4바이트 가변 레코드 길이 제어 필드가 옴을 나타냅니다. 이 레코드 길이 제어 필드는 리틀 엔디안 바이트 순서로 저장됩니다. 이 플래그는 텍스트 제어 문자와는 독립적이며, 텍스트 데이터와 함께 사용되는 경우 레코드의 전체 길이에 모든 제어 문자를 포함합니다. 이 값은 메인프레임 데이터 전송 지원을 위해 제공됩니다. + +#### 4.4.15 external file attributes (4바이트) + +외부 속성의 매핑은 호스트 시스템에 따라 다릅니다('version made by' 참고). MS-DOS의 경우 하위 바이트는 MS-DOS 디렉터리 속성 바이트입니다. 표준 입력에서 입력이 온 경우 이 필드는 0으로 설정됩니다. + +#### 4.4.16 relative offset of local header (4바이트) + +이 파일이 나타나는 첫 번째 디스크의 시작 지점부터 local header가 있어야 하는 위치까지의 오프셋입니다(SHOULD). 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFFFFFF인 경우, 크기는 해당하는 8바이트 zip64 extended information extra field에 담깁니다. + +#### 4.4.17 file name (가변) + +**4.4.17.1** 선택적인 상대 경로를 포함한 파일 이름입니다. 저장되는 경로에는 드라이브 문자나 장치 문자, 또는 선행 슬래시를 포함해서는 안 됩니다(MUST NOT). Amiga 및 UNIX 파일 시스템 등과의 호환성을 위해 모든 슬래시는 역슬래시 '\'가 아닌 정슬래시 '/'여야 합니다(MUST). 표준 입력에서 입력이 온 경우 파일명 필드는 존재하지 않습니다. + +**4.4.17.2** 중앙 디렉터리 암호화 기능을 사용하고 범용 비트 플래그 13이 설정되어 마스킹을 나타내는 경우, 로컬 헤더에 저장된 파일명은 실제 파일명이 아닙니다. 고유한 16진수 값으로 구성된 마스킹 값이 저장됩니다. 이 값은 아카이브 내 각 파일마다 순차적으로 증가합니다. 암호화된 파일명을 조회하는 방법에 대한 자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +#### 4.4.18 file comment (가변) + +이 파일에 대한 코멘트입니다. + +#### 4.4.19 number of this disk (2바이트) + +central directory end record을 포함하는 이 디스크의 번호입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFF인 경우, 크기는 해당하는 4바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.20 number of the disk with the start of the central directory (2바이트) + +중앙 디렉터리가 시작되는 디스크의 번호입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFF인 경우, 크기는 해당하는 4바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.21 total number of entries in the central dir on this disk (2바이트) + +이 디스크에 있는 중앙 디렉터리 항목의 수입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFF인 경우, 크기는 해당하는 8바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.22 total number of entries in the central dir (2바이트) + +.ZIP 파일 내 전체 파일 수입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFF인 경우, 크기는 해당하는 8바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.23 size of the central directory (4바이트) + +전체 중앙 디렉터리의 크기(바이트)입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFFFFFF인 경우, 크기는 해당하는 8바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.24 offset of start of central directory with respect to the starting disk number (4바이트) + +중앙 디렉터리가 시작되는 디스크 상에서 중앙 디렉터리 시작 지점의 오프셋입니다. 아카이브가 ZIP64 형식이고 이 필드의 값이 0xFFFFFFFF인 경우, 크기는 해당하는 8바이트 zip64 end of central directory 필드에 담깁니다. + +#### 4.4.25 .ZIP file comment length (2바이트) + +이 .ZIP 파일에 대한 코멘트의 길이입니다. + +#### 4.4.26 .ZIP file comment (가변) + +이 .ZIP 파일에 대한 코멘트입니다. ZIP 파일 코멘트 데이터는 보호되지 않은 상태로 저장됩니다. 현재로서는 이 영역에 암호화나 데이터 인증이 적용되지 않습니다. 기밀 정보는 이 섹션에 저장해서는 안 됩니다(SHOULD NOT). + +#### 4.4.27 zip64 extensible data sector (가변 크기) + +(현재 PKWARE 전용으로 예약됨) + +#### 4.4.28 extra field (가변) + +이 필드는 저장 공간 확장을 위해 사용해야 합니다(SHOULD). 특수한 애플리케이션 또는 플랫폼 요구를 위해 ZIP 파일 내에 추가 정보를 저장해야 하는 경우, 이 필드에 저장해야 합니다(SHOULD). 이 명세서의 이전 버전을 지원하는 프로그램은 안전하게 이 파일을 건너뛰고 다음 파일이나 헤더를 찾을 수 있습니다. 이 필드는 버전 1.0에서는 길이가 0입니다. + +기존 extra field는 다음의 Extensible data fields 섹션에 정의되어 있습니다. + +### 4.5 확장 가능한 데이터 필드 (Extensible data fields) + +**4.5.1** .ZIP 파일의 'extra' 필드에 서로 다른 프로그램 및 서로 다른 유형의 정보를 저장할 수 있도록, 이 필드에 데이터를 저장하는 모든 프로그램은 반드시 다음 구조를 사용해야 합니다(MUST): + +``` +header1+data1 + header2+data2 . . . +``` + +각 헤더는 반드시 다음으로 구성됩니다(MUST): + +``` +Header ID - 2 bytes +Data Size - 2 bytes +``` + +참고: 모든 필드는 인텔 low-byte/high-byte 순서로 저장됩니다. + +Header ID 필드는 뒤따르는 데이터 블록의 유형을 나타냅니다. + +Header ID 0~31은 PKWARE 전용으로 예약되어 있습니다. 나머지 ID는 서드파티 벤더가 독점적인 용도로 사용할 수 있습니다. + +**4.5.2** PKWARE가 정의한 현재 Header ID 매핑은 다음과 같습니다: + +``` +0x0001 Zip64 extended information extra field +0x0007 AV Info +0x0008 확장 언어 인코딩 데이터(PFS)용으로 예약됨 (APPENDIX D 참고) +0x0009 OS/2 +0x000a NTFS +0x000c OpenVMS +0x000d UNIX +0x000e 파일 스트림 및 fork 디스크립터용으로 예약됨 +0x000f Patch Descriptor +0x0014 X.509 인증서용 PKCS#7 스토어 +0x0015 개별 파일에 대한 X.509 인증서 ID 및 서명 +0x0016 중앙 디렉터리에 대한 X.509 인증서 ID +0x0017 Strong Encryption Header +0x0018 Record Management Controls +0x0019 PKCS#7 암호화 수신자 인증서 목록 +0x0020 타임스탬프 레코드용으로 예약됨 +0x0021 Policy Decryption Key Record +0x0022 Smartcrypt Key Provider Record +0x0023 Smartcrypt Policy Key Data Record +0x0065 IBM S/390(Z390), AS/400(I400) 속성 - 비압축 +0x0066 IBM S/390(Z390), AS/400(I400) 속성 - 압축용으로 예약됨 +0x4690 POSZIP 4690 (예약됨) +``` + +**4.5.3 -Zip64 Extended Information Extra Field (0x0001):** + +다음은 zip64 extended information "extra" 블록의 레이아웃입니다. 로컬 또는 중앙 디렉터리 레코드의 크기 또는 오프셋 필드 중 하나가 필요한 데이터를 담기에 너무 작으면 Zip64 extended information 레코드가 생성됩니다. zip64 extended information 레코드 내 필드의 순서는 고정되어 있지만, 각 필드는 해당하는 로컬 또는 중앙 디렉터리 레코드 필드가 0xFFFF 또는 0xFFFFFFFF로 설정된 경우에만 나타나야 합니다(MUST). + +참고: 모든 필드는 인텔 low-byte/high-byte 순서로 저장됩니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (ZIP64) 0x0001 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| Size | 2바이트 | 이 "extra" 블록의 크기 | +| Original Size | 8바이트 | 원본 비압축 파일 크기 | +| Compressed Size | 8바이트 | 압축 데이터의 크기 | +| Relative Header Offset | 8바이트 | 로컬 헤더 레코드의 오프셋 | +| Disk Start Number | 4바이트 | 이 파일이 시작되는 디스크 번호 | + +로컬 헤더 내 이 항목은 원본 크기와 압축 크기 필드를 모두 반드시 포함해야 합니다(MUST). 중앙 디렉터리를 암호화하고 범용 비트 플래그의 비트 13이 설정되어 마스킹을 나타내는 경우, 로컬 헤더에 저장되는 원본 파일 크기 값은 0이 됩니다. + +**4.5.4 -OS/2 Extra Field (0x0009):** + +다음은 OS/2 속성 "extra" 블록의 레이아웃입니다. (최종 개정: 1995.09.05) + +| 값 | 크기 | 설명 | +|---|---|---| +| (OS/2) 0x0009 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| BSize | 4바이트 | 비압축 블록 크기 | +| CType | 2바이트 | 압축 유형 | +| EACRC | 4바이트 | 비압축 블록에 대한 CRC 값 | +| (var) | 가변 | 압축된 블록 | + +OS/2 확장 속성 구조(FEA2LIST)는 압축된 후 이 구조 전체 안에 저장됩니다. VarFields[]에는 항상 하나의 데이터 "블록"만 존재합니다. + +**4.5.5 -NTFS Extra Field (0x000a):** + +다음은 NTFS 속성 "extra" 블록의 레이아웃입니다. (참고: 현재 Mtime, Atime, Ctime 값은 모든 WIN32 시스템에서 사용될 수 있습니다.) + +| 값 | 크기 | 설명 | +|---|---|---| +| (NTFS) 0x000a | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 전체 "extra" 블록의 크기 | +| Reserved | 4바이트 | 향후 사용을 위해 예약됨 | +| Tag1 | 2바이트 | NTFS 속성 태그 값 #1 | +| Size1 | 2바이트 | 속성 #1의 크기(바이트) | +| (var) | Size1 | 속성 #1 데이터 | +| ... | | | +| TagN | 2바이트 | NTFS 속성 태그 값 #N | +| SizeN | 2바이트 | 속성 #N의 크기(바이트) | +| (var) | SizeN | 속성 #N 데이터 | + +NTFS의 경우 Tag1부터 TagN까지의 값은 다음과 같습니다 (현재 NTFS에는 한 세트의 속성만 정의되어 있음): + +| Tag | 크기 | 설명 | +|---|---|---| +| 0x0001 | 2바이트 | 속성 #1의 태그 | +| Size1 | 2바이트 | 속성 #1의 크기(바이트) | +| Mtime | 8바이트 | 파일 마지막 수정 시간 | +| Atime | 8바이트 | 파일 마지막 접근 시간 | +| Ctime | 8바이트 | 파일 생성 시간 | + +**4.5.6 -OpenVMS Extra Field (0x000c):** + +다음은 OpenVMS 속성 "extra" 블록의 레이아웃입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (VMS) 0x000c | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 전체 "extra" 블록의 크기 | +| CRC | 4바이트 | 블록 나머지 부분에 대한 32비트 CRC | +| Tag1 | 2바이트 | OpenVMS 속성 태그 값 #1 | +| Size1 | 2바이트 | 속성 #1의 크기(바이트) | +| (var) | Size1 | 속성 #1 데이터 | +| ... | | | +| TagN | 2바이트 | OpenVMS 속성 태그 값 #N | +| SizeN | 2바이트 | 속성 #N의 크기(바이트) | +| (var) | SizeN | 속성 #N 데이터 | + +**OpenVMS Extra Field 규칙:** + +**4.5.6.1** 하나 이상의 속성이 존재하며, 각각의 앞에는 위의 TagX 및 SizeX 값이 옵니다. 이 값들은 OpenVMS C의 ATR.H에 정의된 ATR$C_XXXX 및 ATR$S_XXXX 상수와 동일합니다. 이 두 값은 절대 0이 되지 않습니다. + +**4.5.6.2** 워드 정렬이나 패딩은 수행되지 않습니다. + +**4.5.6.3** 잘 작성된 PKZIP/OpenVMS 프로그램은 동일한 TagX 값을 가진 서브 블록을 하나보다 많이 생성하지 않아야 합니다(SHOULD NOT). 또한 특정 디렉터리 레코드에는 유형 0x000c의 "extra" 블록이 하나보다 많이 있어서는 안 됩니다(MUST NOT). + +**4.5.7 -UNIX Extra Field (0x000d):** + +다음은 UNIX "extra" 블록의 레이아웃입니다. 참고: 모든 필드는 인텔 low-byte/high-byte 순서로 저장됩니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (UNIX) 0x000d | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| Atime | 4바이트 | 파일 마지막 접근 시간 | +| Mtime | 4바이트 | 파일 마지막 수정 시간 | +| Uid | 2바이트 | 파일 사용자 ID | +| Gid | 2바이트 | 파일 그룹 ID | +| (var) | 가변 | 가변 길이 데이터 필드 | + +가변 길이 데이터 필드에는 파일 유형별 데이터가 담깁니다. 현재 허용되는 값은 하드/심볼릭 링크에 대한 원본 "링크 대상" 파일명, 그리고 문자/블록 장치 노드에 대한 주(major) 및 부(minor) 장치 노드 번호뿐입니다. 장치 노드는 심볼릭 링크도 하드 링크도 될 수 없으므로 한 세트의 가변 길이 데이터만 저장됩니다. 링크 파일은 원본 파일의 이름을 저장합니다. 이 이름은 NULL로 종료되지 않습니다. 그 크기는 TSize - 12를 확인하여 알 수 있습니다. 장치 항목은 리틀 엔디안 형식의 4바이트 항목 두 개, 총 8바이트로 저장됩니다. 첫 번째 항목은 주 장치 번호, 두 번째는 부 장치 번호입니다. + +**4.5.8 -PATCH Descriptor Extra Field (0x000f):** + +**4.5.8.1** 다음은 Patch Descriptor "extra" 블록의 레이아웃입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (Patch) 0x000f | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 전체 "extra" 블록의 크기 | +| Version | 2바이트 | 디스크립터의 버전 | +| Flags | 4바이트 | 동작 및 반응 (아래 참고) | +| OldSize | 4바이트 | 패치될 파일의 크기 | +| OldCRC | 4바이트 | 패치될 파일의 32비트 CRC | +| NewSize | 4바이트 | 결과 파일의 크기 | +| NewCRC | 4바이트 | 결과 파일의 32비트 CRC | + +**4.5.8.2 동작 및 반응** + +| 비트 | 설명 | +|---|---| +| 0 | 자동 감지에 사용 | +| 1 | self-patch로 취급 | +| 2-3 | 예약됨 | +| 4-5 | 동작(Action, 아래 참고) | +| 6-7 | 예약됨 | +| 8-9 | 파일 없음에 대한 반응(Reaction, 아래 참고) | +| 10-11 | 더 새로운 파일에 대한 반응 | +| 12-13 | 알 수 없는 파일에 대한 반응 | +| 14-15 | 예약됨 | +| 16-31 | 예약됨 | + +**4.5.8.2.1 동작(Actions)** + +| 동작 | 값 | +|---|---| +| none | 0 | +| add | 1 | +| delete | 2 | +| patch | 3 | + +**4.5.8.2.2 반응(Reactions)** + +| 반응 | 값 | +|---|---| +| ask | 0 | +| skip | 1 | +| ignore | 2 | +| fail | 3 | + +**4.5.8.3** 패치 지원은 PKPatchMaker(tm) 기술로 제공되며 미국 특허 및 출원 중인 특허의 적용을 받습니다. 강력 암호화나 패치와 관련된 사항을 포함하여, 현재 APPNOTE에 명시된 특정 기술적 측면을 제품에 사용하거나 구현하려면 PKWARE의 라이선스가 필요합니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.5.9 -PKCS#7 Store for X.509 Certificates (0x0014):** + +이 필드는 파일에 서명하는 데 사용될 수 있는 각 인증서에 대한 정보를 반드시 포함해야 합니다(MUST). ZIP 파일에 대해 중앙 디렉터리 암호화 기능이 활성화된 경우 이 레코드는 Archive Extra Data Record에 나타나며, 그렇지 않으면 첫 번째 중앙 디렉터리 레코드에 나타나고 다른 레코드에서는 무시됩니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (Store) 0x0014 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 스토어 데이터의 크기 | +| TData | TSize | 스토어에 대한 데이터 | + +**4.5.10 -X.509 Certificate ID and Signature for individual file (0x0015):** + +이 필드는 특정 파일에 서명하는 데 PKCS#7 스토어의 어떤 인증서가 사용되었는지에 대한 정보를 포함합니다. 서명 데이터도 포함합니다. 이 필드는 여러 번 나타날 수 있지만 인증서당 한 번만 나타날 수 있습니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (CID) 0x0015 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터의 크기 | +| TData | TSize | 서명 데이터 | + +**4.5.11 -X.509 Certificate ID and Signature for central directory (0x0016):** + +이 필드는 중앙 디렉터리 구조에 서명하는 데 PKCS#7 스토어의 어떤 인증서가 사용되었는지에 대한 정보를 포함합니다. ZIP 파일에 대해 중앙 디렉터리 암호화 기능이 활성화된 경우 이 레코드는 Archive Extra Data Record에 나타나며, 그렇지 않으면 첫 번째 중앙 디렉터리 레코드에 나타납니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (CDID) 0x0016 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터의 크기 | +| TData | TSize | 데이터 | + +**4.5.12 -Strong Encryption Header (0x0017):** + +| 값 | 크기 | 설명 | +|---|---|---| +| 0x0017 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터의 크기 | +| Format | 2바이트 | 이 레코드의 형식 정의 | +| AlgID | 2바이트 | 암호화 알고리즘 식별자 | +| Bitlen | 2바이트 | 암호화 키의 비트 길이 | +| Flags | 2바이트 | 처리 플래그 | +| CertData | TSize-8 | 인증서 복호화용 extra field 데이터 (강력 암호화 명세의 Certificate Processing Method 섹션에서 CertData에 대한 설명 참고) | + +자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.5.13 -Record Management Controls (0x0018):** + +| 값 | 크기 | 설명 | +|---|---|---| +| (Rec-CTL) 0x0018 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| CSize | 2바이트 | 전체 extra 블록 데이터의 크기 | +| Tag1 | 2바이트 | 레코드 제어 속성 1 | +| Size1 | 2바이트 | 속성 1의 크기(바이트) | +| Data1 | Size1 | 속성 1 데이터 | +| ... | | | +| TagN | 2바이트 | 레코드 제어 속성 N | +| SizeN | 2바이트 | 속성 N의 크기(바이트) | +| DataN | SizeN | 속성 N 데이터 | + +**4.5.14 -PKCS#7 Encryption Recipient Certificate List (0x0019):** + +이 필드는 암호화 처리에 사용된 각 인증서에 대한 정보를 포함할 수 있으며(MAY), 암호화된 파일을 복호화할 수 있는 사람을 식별하는 데 사용될 수 있습니다. 이 필드는 archive extra data record에만 나타나야 합니다(SHOULD). 이 필드는 필수는 아니며 공개 암호화 키 데이터를 보존함으로써 아카이브 수정 작업을 돕는 역할만 합니다. 개별 보안 요구사항에 따라 정보 노출을 방지하기 위해 이 데이터를 생략해야 할 수도 있습니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (CStore) 0x0019 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 스토어 데이터의 크기 | +| TData | TSize | 스토어에 대한 데이터 | + +**TData:** + +| 값 | 크기 | 설명 | +|---|---|---| +| Version | 2바이트 | 형식 버전 번호 - 현재는 반드시 0x0001이어야 함 | +| CStore | (var) | PKCS#7 데이터 블롭 | + +자세한 내용은 강력 암호화 명세 섹션을 참고하십시오. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +**4.5.15 -MVS Extra Field (0x0065):** + +다음은 MVS "extra" 블록의 레이아웃입니다. 참고: 일부 필드는 빅 엔디안(Big Endian) 형식으로 저장됩니다. 별도로 명시하지 않는 한 모든 텍스트는 EBCDIC 형식입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (MVS) 0x0065 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| ID | 4바이트 | EBCDIC "Z390" 0xE9F3F9F0 또는 TargetFour용 "T4MV" | +| (var) | TSize-4 | 속성 데이터 (APPENDIX B 참고) | + +**4.5.16 -OS/400 Extra Field (0x0065):** + +다음은 OS/400 "extra" 블록의 레이아웃입니다. 참고: 일부 필드는 빅 엔디안 형식으로 저장됩니다. 별도로 명시하지 않는 한 모든 텍스트는 EBCDIC 형식입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (OS400) 0x0065 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| ID | 4바이트 | EBCDIC "I400" 0xC9F4F0F0 또는 TargetFour용 "T4MV" | +| (var) | TSize-4 | 속성 데이터 (APPENDIX A 참고) | + +**4.5.17 -Policy Decryption Key Record Extra Field (0x0021):** + +다음은 Policy Decryption Key "extra" 블록의 레이아웃입니다. TData는 가변 길이, 가변 내용 필드입니다. 암호화 및/또는 암호화 키 소스에 대한 정보를 담습니다. 현재 TData 구조에 대한 정보는 PKWARE에 문의하십시오. 이 "extra" 블록의 정보는 대신 코멘트 필드에 저장될 수도 있습니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +| 값 | 크기 | 설명 | +|---|---|---| +| 0x0021 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| TData | TSize | 키에 대한 데이터 | + +**4.5.18 -Key Provider Record Extra Field (0x0022):** + +다음은 Key Provider "extra" 블록의 레이아웃입니다. TData는 가변 길이, 가변 내용 필드입니다. 암호화 및/또는 암호화 키 소스에 대한 정보를 담습니다. 현재 TData 구조에 대한 정보는 PKWARE에 문의하십시오. 이 "extra" 블록의 정보는 대신 코멘트 필드에 저장될 수도 있습니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +| 값 | 크기 | 설명 | +|---|---|---| +| 0x0022 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| TData | TSize | 키에 대한 데이터 | + +**4.5.19 -Policy Key Data Record Record Extra Field (0x0023):** + +다음은 Policy Key Data "extra" 블록의 레이아웃입니다. TData는 가변 길이, 가변 내용 필드입니다. 암호화 및/또는 암호화 키 소스에 대한 정보를 담습니다. 현재 TData 구조에 대한 정보는 PKWARE에 문의하십시오. 이 "extra" 블록의 정보는 대신 코멘트 필드에 저장될 수도 있습니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +| 값 | 크기 | 설명 | +|---|---|---| +| 0x0023 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| TSize | 2바이트 | 뒤따르는 데이터 블록의 크기 | +| TData | TSize | 키에 대한 데이터 | + +### 4.6 서드파티 매핑 (Third Party Mappings) + +**4.6.1** 일반적으로 사용되는 서드파티 매핑은 다음과 같습니다: + +``` +0x07c8 Macintosh +0x2605 ZipIt Macintosh +0x2705 ZipIt Macintosh 1.3.5+ +0x2805 ZipIt Macintosh 1.3.5+ +0x334d Info-ZIP Macintosh +0x4341 Acorn/SparkFS +0x4453 Windows NT security descriptor (binary ACL) +0x4704 VM/CMS +0x470f MVS +0x4b46 FWKCS MD5 (아래 참고) +0x4c41 OS/2 access control list (text ACL) +0x4d49 Info-ZIP OpenVMS +0x4f4c Xceed original location extra field +0x5356 AOS/VS (ACL) +0x5455 extended timestamp +0x554e Xceed unicode extra field +0x5855 Info-ZIP UNIX (원본, OS/2, NT 등에도 사용) +0x6375 Info-ZIP Unicode Comment Extra Field +0x6542 BeOS/BeBox +0x7075 Info-ZIP Unicode Path Extra Field +0x756e ASi UNIX +0x7855 Info-ZIP UNIX (신형) +0xa11e Data Stream Alignment (Apache Commons-Compress) +0xa220 Microsoft Open Packaging Growth Hint +0xfd4a SMS/QDOS +0x9901 AE-x 암호화 구조 (APPENDIX E 참고) +0x9902 알 수 없음 +``` + +서드파티 매핑으로 정의된 Extra Field에 대한 상세 설명은 이 데이터 구조에 대한 정보가 PKWARE에 제공되는 대로 문서화될 예정입니다. PKWARE는 발표된 서드파티 데이터의 정확성을 보장하지 않습니다. + +**4.6.2** 서드파티 Extra Field는 본 문서의 Extensible Data Fields 섹션(4.5)에서 정의한 형식을 사용하는 Header ID를 반드시 포함해야 합니다(MUST). + +Data Size 필드는 뒤따르는 데이터 블록의 크기를 나타냅니다. 프로그램은 이 값을 사용하여 관심 없는 데이터 블록을 건너뛰고 다음 헤더 블록으로 이동할 수 있습니다. + +참고: 위에서 언급했듯이, 파일명, 코멘트, extra field를 포함한 전체 .ZIP 파일 헤더의 크기는 64K를 초과하지 않아야 합니다(SHOULD NOT). + +**4.6.3** 서로 다른 두 프로그램이 동일한 Header ID 값을 사용하는 경우를 대비하여, 각 프로그램은 각 데이터 영역의 시작 부분에 최소 2바이트(가급적 4바이트 이상) 크기의 고유한 시그니처를 배치할 것을 강력히 권장합니다. 모든 프로그램은 알려진 유형의 블록이라고 가정하기 전에 Header ID 값이 올바른지 뿐만 아니라 자신의 고유 시그니처가 존재하는지도 확인해야 합니다(SHOULD). + +**서드파티 매핑:** + +**4.6.4 -ZipIt Macintosh Extra Field (long) (0x2605):** + +다음은 Macintosh용 ZipIt extra 블록의 레이아웃입니다. 로컬 헤더 버전과 중앙 헤더 버전은 동일합니다. 파일이 MacBinary로 인코딩되어 저장된 경우 이 블록이 반드시 존재해야 하며(MUST), 파일이 MacBinary로 인코딩되어 있지 않은 경우 사용하지 않아야 합니다(SHOULD NOT). + +| 값 | 크기 | 설명 | +|---|---|---| +| (Mac2) 0x2605 | Short | 이 extra 블록 유형의 태그 | +| TSize | Short | 이 블록의 전체 데이터 크기 | +| "ZPIT" | beLong | extra-field 시그니처 | +| FnLen | Byte | FileName의 길이 | +| FileName | 가변 | 전체 Macintosh 파일명 | +| FileType | Byte[4] | 4바이트 Mac 파일 유형 문자열 | +| Creator | Byte[4] | 4바이트 Mac creator 문자열 | + +**4.6.5 -ZipIt Macintosh Extra Field (short, for files) (0x2705):** + +다음은 Macintosh용 ZipIt extra 블록의 축약형("full name" 항목 없음) 레이아웃입니다. 이 변형은 ZipIt 1.3.5 이상에서 MacBinary로 인코딩되지 않은 파일(디렉터리 아님) 항목에 사용됩니다. 로컬 헤더 버전과 중앙 헤더 버전은 동일합니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (Mac2b) 0x2705 | Short | 이 extra 블록 유형의 태그 | +| TSize | Short | 이 블록의 전체 데이터 크기 (12) | +| "ZPIT" | beLong | extra-field 시그니처 | +| FileType | Byte[4] | 4바이트 Mac 파일 유형 문자열 | +| Creator | Byte[4] | 4바이트 Mac creator 문자열 | +| fdFlags | beShort | FInfo.frFlags의 속성, 생략 가능(MAY) | +| 0x0000 | beShort | 예약됨, 생략 가능(MAY) | + +**4.6.6 -ZipIt Macintosh Extra Field (short, for directories) (0x2805):** + +다음은 디렉터리 항목에만 사용되는 Macintosh용 ZipIt extra 블록 축약형의 레이아웃입니다. 이 변형은 ZipIt 1.3.5 이상에서 디렉터리에 대한 선택적인 Mac 관련 정보를 저장하는 데 사용됩니다. 로컬 헤더 버전과 중앙 헤더 버전은 동일합니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (Mac2c) 0x2805 | Short | 이 extra 블록 유형의 태그 | +| TSize | Short | 이 블록의 전체 데이터 크기 (12) | +| "ZPIT" | beLong | extra-field 시그니처 | +| frFlags | beShort | DInfo.frFlags의 속성, 생략 가능(MAY) | +| View | beShort | ZipIt view 플래그, 생략 가능(MAY) | + +View 필드는 다음과 같이 ZipIt 내부 설정을 지정합니다: + +플래그의 비트: +- 비트 0: 설정된 경우, ZipIt에서 아카이브 내용을 볼 때 폴더가 펼쳐진(열린) 상태로 표시됩니다. +- 비트 1-15: 예약됨, 0 + +**4.6.7 -FWKCS MD5 Extra Field (0x4b46):** + +파일명과 무관하게 파일을 자동으로 식별하는 데 사용되는 FWKCS Contents_Signature System은 선택적으로 extra field를 추가하고 사용하여 향상된 contents_signature의 신속한 생성을 지원합니다: + +``` +Header ID = 0x4b46 +Data Size = 0x0013 +Preface = 'M','D','5' +뒤이어 비압축 파일의 128비트 MD5 해시(1)가 16바이트로 오며, +낮은 바이트가 먼저 옵니다. +``` + +FWKCS가 이 extra field를 추가하기 위해 .ZIP 파일 중앙 디렉터리를 개정할 때, 해당 파일의 비압축 파일 길이에 대한 중앙 디렉터리 항목도 실측값으로 교체합니다. + +FWKCS는 .ZIP 파일 중앙 디렉터리에서 이 extra field가 있는 경우 이를 제거하는 옵션을 제공합니다. 이 extra field를 추가할 때 FWKCS는 .ZIP 파일 Authenticity Verification을 보존하며, 이 extra field를 제거할 때는 PKZIP 버전 2.04g까지의 모든 AV 버전을 보존합니다. + +FWKCS 및 FWKCS Contents_Signature System은 Frederick W. Kantor의 상표입니다. + +(1) R. Rivest, RFC1321.TXT, MIT Laboratory for Computer Science and RSA Data Security, Inc., 1992년 4월. 76-77행: "MD5 알고리즘은 검토 및 표준으로서의 채택 가능성을 위해 퍼블릭 도메인으로 공개됩니다." + +**4.6.8 -Info-ZIP Unicode Comment Extra Field (0x6375):** + +중앙 디렉터리 헤더에 저장된 파일 코멘트의 UTF-8 버전을 저장합니다. (최종 개정: 20070912) + +| 값 | 크기 | 설명 | +|---|---|---| +| (UCom) 0x6375 | Short | 이 extra 블록 유형의 태그 ("uc") | +| TSize | Short | 이 블록의 전체 데이터 크기 | +| Version | 1바이트 | 이 extra field의 버전, 현재는 1 | +| ComCRC32 | 4바이트 | 코멘트 필드의 CRC32 체크섬 | +| UnicodeCom | 가변 | 항목 코멘트의 UTF-8 버전 | + +현재 Version은 숫자 1로 설정되어 있습니다. 이 필드를 변경할 필요가 있는 경우 버전이 증가됩니다. 변경 사항은 하위 호환되지 않을 수 있으므로(MAY NOT), 버전을 인식할 수 없는 경우 이 extra field를 사용하지 않아야 합니다(SHOULD NOT). + +ComCRC32는 중앙 디렉터리 헤더의 File Comment 필드에 대한 표준 zip CRC32 체크섬입니다. 이는 Unicode Comment extra field가 생성된 이후 코멘트 필드가 변경되지 않았는지 확인하는 데 사용됩니다. 유틸리티가 File Comment 필드를 변경하면서 UTF-8 Comment extra field를 갱신하지 않으면 이러한 상황이 발생할 수 있습니다. CRC 검사에 실패하면 이 Unicode Comment extra field는 무시해야 하며(SHOULD), 대신 헤더의 File Comment 필드를 사용해야 합니다(SHOULD). + +UnicodeCom 필드는 헤더 내 File Comment 필드의 UTF-8 버전입니다. UnicodeCom은 UTF-8로 정의되므로 UTF-8 바이트 순서 표시(BOM)는 사용되지 않습니다. 이 필드의 길이는 TSize에서 이전 필드들의 크기를 뺀 값으로 결정됩니다. File Name과 Comment 필드가 모두 UTF-8인 경우, 새로운 범용 비트 플래그의 비트 11(언어 인코딩 플래그, EFS)을 사용하여 헤더의 File Name과 Comment 필드가 모두 UTF-8임을 나타낼 수 있으며, 이 경우 Unicode Path 및 Unicode Comment extra field는 필요하지 않으므로 생성하지 않아야 합니다(SHOULD NOT). 하위 호환성을 위해, 비트 11은 압축 대상 경로와 코멘트의 원본 문자 집합이 이미 UTF-8인 경우에만 사용해야 함(SHOULD)에 유의하십시오. 파일에 대해 로컬 헤더와 중앙 디렉터리 헤더 모두 동일한 파일 코멘트 저장 방식(범용 비트 11 또는 extra field)이 사용될 것으로 기대됩니다. + +**4.6.9 -Info-ZIP Unicode Path Extra Field (0x7075):** + +로컬 헤더와 중앙 디렉터리 헤더에 저장된 파일명 필드의 UTF-8 버전을 저장합니다. (최종 개정: 20070912) + +| 값 | 크기 | 설명 | +|---|---|---| +| (UPath) 0x7075 | Short | 이 extra 블록 유형의 태그 ("up") | +| TSize | Short | 이 블록의 전체 데이터 크기 | +| Version | 1바이트 | 이 extra field의 버전, 현재는 1 | +| NameCRC32 | 4바이트 | 파일명 필드의 CRC32 체크섬 | +| UnicodeName | 가변 | 항목 파일명의 UTF-8 버전 | + +현재 Version은 숫자 1로 설정되어 있습니다. 이 필드를 변경할 필요가 있는 경우 버전이 증가됩니다. 변경 사항은 하위 호환되지 않을 수 있으므로(MAY NOT), 버전을 인식할 수 없는 경우 이 extra field를 사용하지 않아야 합니다(SHOULD NOT). + +NameCRC32는 헤더의 File Name 필드에 대한 표준 zip CRC32 체크섬입니다. 이는 Unicode Path extra field가 생성된 이후 헤더의 File Name 필드가 변경되지 않았는지 확인하는 데 사용됩니다. 유틸리티가 File Name을 변경하면서 UTF-8 path extra field를 갱신하지 않으면 이러한 상황이 발생할 수 있습니다. CRC 검사에 실패하면 이 UTF-8 Path Extra Field는 무시해야 하며(SHOULD), 대신 헤더의 File Name 필드를 사용해야 합니다(SHOULD). + +UnicodeName은 헤더 내 File Name 필드 내용의 UTF-8 버전입니다. UnicodeName은 UTF-8로 정의되므로 UTF-8 바이트 순서 표시(BOM)는 사용되지 않습니다. 이 필드의 길이는 TSize에서 이전 필드들의 크기를 뺀 값으로 결정됩니다. File Name과 Comment 필드가 모두 UTF-8인 경우, 새로운 범용 비트 플래그의 비트 11(언어 인코딩 플래그, EFS)을 사용하여 헤더의 File Name과 Comment 필드가 모두 UTF-8임을 나타낼 수 있으며, 이 경우 Unicode Path 및 Unicode Comment extra field는 필요하지 않으므로 생성하지 않아야 합니다(SHOULD NOT). 하위 호환성을 위해, 비트 11은 압축 대상 경로와 코멘트의 원본 문자 집합이 이미 UTF-8인 경우에만 사용해야 함(SHOULD)에 유의하십시오. 파일에 대해 로컬 헤더와 중앙 디렉터리 헤더 모두 동일한 파일명 저장 방식(범용 비트 11 또는 extra field)이 사용될 것으로 기대됩니다. + +**4.6.10 -Microsoft Open Packaging Growth Hint (0xa220):** + +| 값 | 크기 | 설명 | +|---|---|---| +| 0xa220 | Short | 이 extra 블록 유형의 태그 | +| TSize | Short | Sig + PadVal + Padding의 크기 | +| Sig | Short | 검증 시그니처 (A028) | +| PadVal | Short | 초기 패딩 값 | +| Padding | 가변 | NULL 문자로 채워짐 | + +**4.6.11 -Data Stream Alignment (Apache Commons-Compress) (0xa11e):** + +(Zbynek Vyskovsky 제공) ZIP 아카이브 내에서 해당 항목의 데이터 스트림 정렬(alignment)을 정의합니다. 또한 ZIP 파일을 재압축할 때 압축 방법을 유지해야 하는지를 나타냅니다. + +이 extra field의 목적은 특정 리소스를 워드 또는 페이지 경계에 정렬시켜 메모리에 쉽게 매핑할 수 있도록 하는 것입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| 0xa11e | Short | 이 extra 블록 유형의 태그 | +| TSize | Short | 이 블록의 전체 데이터 크기 (2+padding) | +| alignment | Short | 요구되는 정렬 및 지시자 | +| 0x00 | 가변 | 패딩 | + +alignment 필드(하위 15비트)는 데이터 스트림에 필요한 최소 정렬을 정의합니다. alignment 필드의 비트 15는 ZIP 파일을 재압축할 때 이 항목의 압축 방법을 변경할 수 있는지를 나타냅니다. 값 0은 압축 방법을 변경해서는 안 됨을, 값 1은 압축 방법을 변경할 수 있음을 나타냅니다. padding 필드는 올바른 정렬을 보장하기 위한 패딩을 포함합니다. 오프셋이나 요구되는 정렬이 변경될 때는 언제든지 이 값을 변경할 수 있습니다. (https://issues.apache.org/jira/browse/COMPRESS-391 참고) + +### 4.7 매니페스트 파일 + +**4.7.1** ZIP 파일을 사용하는 애플리케이션은 ZIP 파일에 넣는 파일과 함께 포함되어야 하는 추가 정보가 필요할 수 있습니다(MAY). 정의된 ZIP 저장 레코드로 저장할 수 없는 애플리케이션별 정보는 본 문서에 정의된 확장 가능한 Extra Field 규약을 사용하여 저장해야 합니다(SHOULD). 다만 일부 애플리케이션은 추가 정보를 저장하는 수단으로 매니페스트 파일을 사용할 수 있습니다(MAY). 한 예로 .JAR 확장자를 가진 ZIP 형식 파일(JAR 파일)에서 사용되는 META-INF/MANIFEST.MF 파일이 있습니다. + +**4.7.2** 매니페스트 파일은 이 정보를 필요로 하는 애플리케이션 프로세스를 위해 생성된 파일입니다. 매니페스트 파일은 이를 정의하는 애플리케이션 프로세스가 요구하는 어떤 파일 유형이든 될 수 있습니다(MAY). 이 정보가 적용되는 파일들과 동일한 ZIP 파일 안에 배치됩니다. 관례적으로 이 파일은 일반적으로 ZIP 파일에 배치되는 첫 번째 파일이며, 정의된 디렉터리 경로를 포함할 수 있습니다(MAY). + +**4.7.3** 매니페스트 파일은 ZIP 파일 내 파일들에 대한 애플리케이션 처리 요구에 따라 압축되거나 암호화될 수 있습니다(MAY). + +매니페스트 파일은 본 명세서의 범위를 벗어납니다. + +--- + +## 5.0 압축 방법 설명 + +### 5.1 UnShrinking - 방법 1 + +**5.1.1** Shrinking은 부분 클리어링을 사용하는 동적 Ziv-Lempel-Welch 압축 알고리즘입니다. 초기 코드 크기는 9비트이며 최대 코드 크기는 13비트입니다. Shrinking은 다음과 같은 몇 가지 측면에서 일반적인 동적 Ziv-Lempel-Welch 구현과 다릅니다: + +**5.1.2** 코드 크기는 압축기(compressor)가 제어하며, 현재 코드 크기보다 큰 코드가 생성되더라도(반드시 사용되지는 않더라도) 자동으로 증가하지 않습니다. 압축 해제기(decompressor)는 코드 시퀀스 256(십진수) 다음에 1이 오는 것을 만나면, 입력 스트림에서 읽는 코드 크기를 다음 비트 크기로 늘려야 합니다(SHOULD). 코드에 대한 블로킹은 수행되지 않으므로, 증가된 크기의 다음 코드는 이전의 더 작은 비트 크기로 읽었던 지점 바로 다음에서 입력 스트림으로부터 읽어야 합니다(SHOULD). 다시 말해, 압축 해제기는 시퀀스 256,1을 만나기 전까지는 사용 중인 코드 크기를 늘려서는 안 됩니다(SHOULD NOT). + +**5.1.3** 테이블이 가득 차면 전체 클리어링은 수행되지 않습니다. 대신 압축기가 코드 시퀀스 256,2(십진수)를 내보내면, 압축 해제기는 Ziv-Lempel 트리에서 모든 리프 노드를 지우고(SHOULD) 현재 코드 크기를 계속 사용해야 합니다. Ziv-Lempel 트리에서 지워진 노드들은 이후 재사용되며, 가장 낮은 코드 값이 먼저 재사용되고 가장 높은 코드 값이 마지막에 재사용됩니다. 압축기는 언제든지 시퀀스 256,2를 내보낼 수 있습니다. + +### 5.2 Expanding - 방법 2-5 + +**5.2.1** Reducing 알고리즘은 실제로는 두 가지 별개 알고리즘의 조합입니다. 첫 번째 알고리즘은 반복되는 바이트 시퀀스를 압축하고, 두 번째 알고리즘은 첫 번째 알고리즘의 압축된 스트림에 확률적 압축 방법을 적용합니다. + +**5.2.2** 확률적 압축은 가능한 각 ASCII 문자에 대응하는 j=0부터 255까지의 'follower set' S(j) 배열을 저장합니다. 각 세트는 0개에서 32개 사이의 문자를 포함하며, S(j)[0],...,S(j)[m] (m<32)로 표기합니다. 이 세트들은 Reduced 파일의 데이터 영역 시작 부분에 역순으로 저장되며, S(255)가 먼저, S(0)이 마지막입니다. + +**5.2.3** 각 세트는 { N(j), S(j)[0],...,S(j)[N(j)-1] } 형태로 인코딩되며, 여기서 N(j)는 세트 S(j)의 크기입니다. N(j)는 0일 수 있으며, 이 경우 S(j)의 follower set은 비어 있습니다. 각 N(j) 값은 6비트로 인코딩되며, 그 뒤에 S(j)[0]부터 S(j)[N(j)-1]에 각각 대응하는 N(j)개의 8비트 문자 값이 옵니다. N(j)가 0이면 S(j)에 대한 값은 저장되지 않고, N(j-1) 값이 바로 뒤따릅니다. + +**5.2.4** follower set 바로 뒤에는 압축된 데이터 스트림이 옵니다. 확률적 압축 해제를 위해 압축된 데이터 스트림은 다음과 같이 해석될 수 있습니다: + +``` +Last-Character <- 0로 둔다. +완료될 때까지 반복 + follower set S(Last-Character)가 비어 있으면 + 입력 스트림에서 8비트를 읽어 출력 스트림에 그 값을 복사한다. + 그렇지 않고 follower set S(Last-Character)가 비어 있지 않으면 + 입력 스트림에서 1비트를 읽는다. + 이 비트가 0이 아니면 + 입력 스트림에서 8비트를 읽어 출력 스트림에 그 값을 복사한다. + 그렇지 않고 이 비트가 0이면 + 입력 스트림에서 B(N(Last-Character))비트를 읽어 I에 대입한다. + S(Last-Character)[I]의 값을 출력 스트림에 복사한다. + + 출력 스트림에 마지막으로 놓인 값을 Last-Character에 대입한다. +반복 종료 +``` + +B(N(j))는 값 N(j)-1을 인코딩하는 데 필요한 최소 비트 수로 정의됩니다. + +**5.2.5** 위에서 얻은 압축 해제된 스트림은 다음과 같이 확장하여 원본 파일을 재생성할 수 있습니다: + +``` +State <- 0으로 둔다. + +완료될 때까지 반복 + 입력 스트림에서 8비트를 읽어 C에 넣는다. + State에 따라 분기: + 0: C가 DLE(십진수 144)이 아니면 + C를 출력 스트림에 복사한다. + 그렇지 않고 C가 DLE이면 + State <- 1로 둔다. + + 1: C가 0이 아니면 + V <- C로 둔다. + Len <- L(V)로 둔다. + State <- F(Len)으로 둔다. + 그렇지 않고 C가 0이면 + 값 144(십진수)를 출력 스트림에 복사한다. + State <- 0으로 둔다. + + 2: Len <- Len + C로 둔다. + State <- 3으로 둔다. + + 3: 출력 스트림에서 D(V,C)바이트만큼 뒤로 이동한다. + (이 위치가 출력 스트림의 시작보다 앞이면, + 출력 스트림 시작 이전의 모든 데이터는 + 0으로 채워져 있다고 가정한다.) + 이 위치에서 Len+3바이트를 출력 스트림에 복사한다. + State <- 0으로 둔다. + 분기 종료 +반복 종료 +``` + +함수 F, L, D는 '압축 계수'(1~4)에 따라 다음과 같이 정의됩니다: + +``` +압축 계수 1의 경우: + L(X)는 X의 하위 7비트와 같다. + F(X)는 X가 127이면 2, 그렇지 않으면 3이다. + D(X,Y)는 (X의 상위 1비트) * 256 + Y + 1이다. +압축 계수 2의 경우: + L(X)는 X의 하위 6비트와 같다. + F(X)는 X가 63이면 2, 그렇지 않으면 3이다. + D(X,Y)는 (X의 상위 2비트) * 256 + Y + 1이다. +압축 계수 3의 경우: + L(X)는 X의 하위 5비트와 같다. + F(X)는 X가 31이면 2, 그렇지 않으면 3이다. + D(X,Y)는 (X의 상위 3비트) * 256 + Y + 1이다. +압축 계수 4의 경우: + L(X)는 X의 하위 4비트와 같다. + F(X)는 X가 15이면 2, 그렇지 않으면 3이다. + D(X,Y)는 (X의 상위 4비트) * 256 + Y + 1이다. +``` + +### 5.3 Imploding - 방법 6 + +**5.3.1** Imploding 알고리즘은 실제로는 두 가지 별개 알고리즘의 조합입니다. 첫 번째 알고리즘은 슬라이딩 딕셔너리를 사용하여 반복되는 바이트 시퀀스를 압축합니다. 두 번째 알고리즘은 여러 개의 Shannon-Fano 트리를 사용하여 슬라이딩 딕셔너리 출력의 인코딩을 압축하는 데 사용됩니다. + +**5.3.2** Imploding 알고리즘은 4K 또는 8K 슬라이딩 딕셔너리 크기를 사용할 수 있습니다. 사용되는 딕셔너리 크기는 범용 플래그 워드의 비트 1로 판단할 수 있습니다. 0비트는 4K 딕셔너리, 1비트는 8K 딕셔너리를 나타냅니다. + +**5.3.3** Shannon-Fano 트리는 압축된 파일의 시작 부분에 저장됩니다. 저장되는 트리의 개수는 범용 플래그 워드의 비트 2로 정의됩니다. 0비트는 2개의 트리가 저장됨을, 1비트는 3개의 트리가 저장됨을 나타냅니다. 3개의 트리가 저장된 경우, 첫 번째 Shannon-Fano 트리는 리터럴 문자의 인코딩을, 두 번째 트리는 길이(Length) 정보의 인코딩을, 세 번째 트리는 거리(Distance) 정보의 인코딩을 나타냅니다. 2개의 Shannon-Fano 트리가 저장된 경우, Length 트리가 먼저 저장되고 그 뒤에 Distance 트리가 저장됩니다. + +**5.3.4** 존재하는 경우 Literal Shannon-Fano 트리는 전체 ASCII 문자 집합을 나타내는 데 사용되며 256개의 값을 포함합니다. 이 트리는 슬라이딩 딕셔너리 알고리즘으로 압축되지 않은 모든 데이터를 압축하는 데 사용됩니다. 이 트리가 존재하는 경우, 슬라이딩 딕셔너리의 최소 일치 길이(Minimum Match Length)는 3입니다. 이 트리가 존재하지 않는 경우 최소 일치 길이는 2입니다. + +**5.3.5** Length Shannon-Fano 트리는 슬라이딩 딕셔너리 출력에서 나온 (length, distance) 쌍의 Length 부분을 압축하는 데 사용됩니다. Length 트리는 최소 일치 길이부터 최소 일치 길이+63까지의 범위를 갖는 64개의 값을 포함합니다. + +**5.3.6** Distance Shannon-Fano 트리는 슬라이딩 딕셔너리 출력에서 나온 (length, distance) 쌍의 Distance 부분을 압축하는 데 사용됩니다. Distance 트리는 0부터 63까지의 범위를 갖는 64개의 값을 포함하며, 이는 distance 값의 상위 6비트를 나타냅니다. distance 값 자체는 0에서 슬라이딩 딕셔너리 크기(4K 또는 8K) 사이의 값입니다. + +**5.3.7** Shannon-Fano 트리 자체는 압축된 형식으로 저장됩니다. 트리 데이터의 첫 번째 바이트는 (압축된) Shannon-Fano 트리를 나타내는 데이터의 바이트 수에서 1을 뺀 값을 나타냅니다. 나머지 바이트는 다음과 같이 인코딩된 Shannon-Fano 트리 데이터를 나타냅니다: + +``` +상위 4비트: 이 비트 길이를 갖는 값의 개수 + 1 (1 - 16) +하위 4비트: 값을 나타내는 데 필요한 비트 길이 + 1 (1 - 16) +``` + +**5.3.8** Shannon-Fano 코드는 다음 알고리즘을 사용하여 비트 길이로부터 구성할 수 있습니다: + +``` +1) 파일에 저장된 원래 길이의 순서를 유지한 채, 비트 길이를 오름차순으로 정렬한다. + +2) Shannon-Fano 트리를 생성한다: + + Code <- 0 + CodeIncrement <- 0 + LastBitLength <- 0 + i <- Shannon-Fano 코드 개수 - 1 (255 또는 63) + + i >= 0인 동안 반복 + Code = Code + CodeIncrement + BitLength(i) <> LastBitLength이면 + LastBitLength=BitLength(i) + CodeIncrement = 1을 (16 - LastBitLength)만큼 왼쪽 시프트 + ShannonCode(i) = Code + i <- i - 1 + 반복 종료 + +3) 위 ShannonCode() 벡터의 모든 비트 순서를 반전시켜, 최상위 비트가 최하위 + 비트가 되도록 한다. 예를 들어 값 0x1234(16진수)는 0x2C48(16진수)이 된다. + +4) Shannon-Fano 코드의 순서를 파일에 원래 저장된 순서대로 복원한다. +``` + +**예시:** + +크기 8의 Shannon-Fano 트리 인코딩 예시입니다. Imploding에 실제 사용되는 Shannon-Fano 트리는 64개 또는 256개 항목 크기임에 유의하십시오. + +예시: `0x02, 0x42, 0x01, 0x13` + +첫 번째 바이트는 이 테이블에 3개의 값이 있음을 나타냅니다. 바이트를 디코딩하면: + +``` +0x42 = 3비트 길이 코드 5개 +0x01 = 2비트 길이 코드 1개 +0x13 = 4비트 길이 코드 2개 +``` + +이는 다음의 원본 비트 길이 배열을 생성합니다: `(3, 3, 3, 3, 3, 2, 4, 4)` + +이 테이블에는 값 0부터 7까지 8개의 코드가 있습니다. 알고리즘을 사용하여 Shannon-Fano 코드를 구하면: + +| Val | 정렬 | 구성된 코드 | 반전된 값 | 순서 복원 | 원본 길이 | +|---|---|---|---|---|---| +| 0: | 2 | 1100000000000000 | 11 | 101 | 3 | +| 1: | 3 | 1010000000000000 | 101 | 001 | 3 | +| 2: | 3 | 1000000000000000 | 001 | 110 | 3 | +| 3: | 3 | 0110000000000000 | 110 | 010 | 3 | +| 4: | 3 | 0100000000000000 | 010 | 100 | 3 | +| 5: | 3 | 0010000000000000 | 100 | 11 | 2 | +| 6: | 4 | 0001000000000000 | 1000 | 1000 | 4 | +| 7: | 4 | 0000000000000000 | 0000 | 0000 | 4 | + +Val, Order Restored, Original Length 열의 값들이 이제 Shannon-Fano 인코딩된 데이터를 디코딩하는 데 사용할 수 있는 Shannon-Fano 인코딩 트리를 나타냅니다. 데이터 스트림에서 가변 길이의 Shannon-Fano 값을 파싱하는 방법은 본 문서의 범위를 벗어납니다(자세한 내용은 본 문서 끝의 참고문헌 목록 참고). 다만 Greenlaw 알고리즘과 같이 Huffman 가변 길이 디코딩에 사용되는 전통적인 디코딩 기법을 성공적으로 적용할 수 있습니다. + +**5.3.9** 압축된 데이터 스트림은 압축된 Shannon-Fano 데이터 바로 뒤에서 시작됩니다. 압축된 데이터 스트림은 다음과 같이 해석될 수 있습니다: + +``` +완료될 때까지 반복 + 입력 스트림에서 1비트를 읽는다. + + 이 비트가 0이 아니면 (인코딩된 데이터는 리터럴 데이터) + Literal Shannon-Fano 트리가 존재하면 + Literal Shannon-Fano 트리를 사용해 문자를 읽고 디코딩한다. + 그렇지 않으면 + 입력 스트림에서 8비트를 읽는다. + 문자를 출력 스트림에 복사한다. + 그렇지 않으면 (인코딩된 데이터는 슬라이딩 딕셔너리 일치) + 8K 딕셔너리 크기이면 + 오프셋 Distance를 위해 7비트를 읽는다 (오프셋의 하위 7비트). + 그렇지 않으면 + 오프셋 Distance를 위해 6비트를 읽는다 (오프셋의 하위 6비트). + + Distance Shannon-Fano 트리를 사용하여 Distance 값의 상위 6비트를 + 읽고 디코딩한다. + + Length Shannon-Fano 트리를 사용하여 Length 값을 읽고 디코딩한다. + + Length <- Length + 최소 일치 길이 + + Length = 63 + 최소 일치 길이이면 + 입력 스트림에서 8비트를 읽어 Length에 더한다. + + 출력 스트림에서 Distance+1바이트만큼 뒤로 이동한 후, 이 위치에서 + Length개의 문자를 출력 스트림에 복사한다. (이 위치가 출력 스트림의 + 시작보다 앞이면, 출력 스트림 시작 이전의 모든 데이터는 0으로 + 채워져 있다고 가정한다.) +반복 종료 +``` + +### 5.4 Tokenizing - 방법 7 + +**5.4.1** 이 방법은 PKZIP에서 사용되지 않습니다. + +### 5.5 Deflating - 방법 8 + +**5.5.1** Deflate 알고리즘은 최대 32K의 슬라이딩 딕셔너리를 사용하고 Huffman/Shannon-Fano 코드로 2차 압축을 수행한다는 점에서 Implode 알고리즘과 유사합니다. + +**5.5.2** 압축된 데이터는 블록을 설명하는 헤더 및 데이터 블록에서 사용된 Huffman 코드와 함께 블록 단위로 저장됩니다. 헤더 형식은 다음과 같습니다: + +``` +비트 0: Last Block 비트 이 비트는 데이터 내 마지막 압축 블록인 경우 1로 설정됩니다. +비트 1-2: 블록 유형 + 00 (0) - 블록이 저장됨(stored) - 모든 저장된 데이터는 바이트 정렬됩니다. + 다음 바이트까지 비트를 건너뛴 후, 다음 워드가 블록 길이가 되며, + 뒤이어 블록 길이 워드의 1의 보수가 옵니다. 블록의 나머지 데이터는 + 저장된 데이터입니다. + + 01 (1) - 리터럴 및 거리 코드에 고정 Huffman 코드를 사용합니다. + Lit Code 비트 Dist Code 비트 + --------- ---- --------- ---- + 0 - 143 8 0 - 31 5 + 144 - 255 9 + 256 - 279 7 + 280 - 287 8 + + 리터럴 코드 286-287과 거리 코드 30-31은 사용되지 않지만 + huffman 구성에는 참여합니다. + + 10 (2) - 동적(Dynamic) Huffman 코드. (Huffman 코드 확장 참고) + + 11 (3) - 예약됨 - 발견되면 "압축 데이터 오류" 플래그를 표시합니다. +``` + +**5.5.3 Huffman 코드 확장** + +데이터 블록이 동적 Huffman 코드로 저장된 경우, Huffman 코드는 다음과 같은 압축 형식으로 전송됩니다: + +``` +5비트: 전송된 Literal 코드 수 - 256 (256 - 286) + 그 밖의 모든 코드는 전송되지 않습니다. +5비트: Dist 코드 수 - 1 (1 - 32) +4비트: Bit Length 코드 수 - 3 (3 - 19) +``` + +Huffman 코드는 비트 길이로 전송되며, Implode 알고리즘에서 설명한 대로 코드가 구성됩니다. 비트 길이 자체는 Huffman 코드로 압축됩니다. 19개의 비트 길이 코드가 있습니다: + +``` +0 - 15: 비트 길이 0 - 15를 나타냅니다. + 16: 이전 비트 길이를 3 - 6회 반복 복사합니다. + 다음 2비트는 반복 길이를 나타냅니다 (0 = 3, ... , 3 = 6) + 예시: 코드 8, 16(+2비트 11), 16(+2비트 10)은 + 비트 길이 8인 12개(1 + 6 + 5)로 확장됩니다. + 17: 비트 길이 0을 3 - 10회 반복합니다. (3비트 길이) + 18: 비트 길이 0을 11 - 138회 반복합니다. (7비트 길이) +``` + +비트 길이 코드의 길이는 다음 순서로 값당 3비트(0 - 7)씩 압축되어 전송됩니다: + +``` +16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2, 14, 1, 15 +``` + +Huffman 코드는 Implode 알고리즘에서 설명한 방식으로 구성해야 하지만(SHOULD), 가장 짧은 비트 길이부터 코드가 할당된다는 점이 다릅니다. 즉, 가장 짧은 코드는 모두 1이 아니라 모두 0이어야 합니다(SHOULD). 또한 비트 길이가 0인 코드는 트리 구성에 참여하지 않습니다. 이 코드들은 이후 리터럴 및 거리 테이블의 비트 길이를 디코딩하는 데 사용됩니다. + +리터럴 테이블의 비트 길이는 앞서 전송된 5비트로 설명된 항목 수만큼 먼저 전송됩니다. 최대 286개의 리터럴 문자가 있으며, 처음 256개는 각각의 8비트 문자를 나타내고, 코드 256은 블록 종료(End-Of-Block) 코드를 나타내며, 나머지 29개 코드는 3부터 258까지의 복사 길이를 나타냅니다. 아래 설명된 대로 1부터 32k까지의 거리를 나타내는 최대 30개의 거리 코드가 있습니다. + +**Length Codes (길이 코드)** + +| Code | Extra Bits | Length | Code | Extra Bits | Lengths | Code | Extra Bits | Lengths | Code | Extra Bits | Length(s) | +|---|---|---|---|---|---|---|---|---|---|---|---| +| 257 | 0 | 3 | 265 | 1 | 11,12 | 273 | 3 | 35-42 | 281 | 5 | 131-162 | +| 258 | 0 | 4 | 266 | 1 | 13,14 | 274 | 3 | 43-50 | 282 | 5 | 163-194 | +| 259 | 0 | 5 | 267 | 1 | 15,16 | 275 | 3 | 51-58 | 283 | 5 | 195-226 | +| 260 | 0 | 6 | 268 | 1 | 17,18 | 276 | 3 | 59-66 | 284 | 5 | 227-257 | +| 261 | 0 | 7 | 269 | 2 | 19-22 | 277 | 4 | 67-82 | 285 | 0 | 258 | +| 262 | 0 | 8 | 270 | 2 | 23-26 | 278 | 4 | 83-98 | | | | +| 263 | 0 | 9 | 271 | 2 | 27-30 | 279 | 4 | 99-114 | | | | +| 264 | 0 | 10 | 272 | 2 | 31-34 | 280 | 4 | 115-130 | | | | + +**Distance Codes (거리 코드)** + +| Code | Extra Bits | Dist | Code | Extra Bits | Dist | Code | Extra Bits | Distance | Code | Extra Bits | Distance | +|---|---|---|---|---|---|---|---|---|---|---|---| +| 0 | 0 | 1 | 8 | 3 | 17-24 | 16 | 7 | 257-384 | 24 | 11 | 4097-6144 | +| 1 | 0 | 2 | 9 | 3 | 25-32 | 17 | 7 | 385-512 | 25 | 11 | 6145-8192 | +| 2 | 0 | 3 | 10 | 4 | 33-48 | 18 | 8 | 513-768 | 26 | 12 | 8193-12288 | +| 3 | 0 | 4 | 11 | 4 | 49-64 | 19 | 8 | 769-1024 | 27 | 12 | 12289-16384 | +| 4 | 1 | 5,6 | 12 | 5 | 65-96 | 20 | 9 | 1025-1536 | 28 | 13 | 16385-24576 | +| 5 | 1 | 7,8 | 13 | 5 | 97-128 | 21 | 9 | 1537-2048 | 29 | 13 | 24577-32768 | +| 6 | 2 | 9-12 | 14 | 6 | 129-192 | 22 | 10 | 2049-3072 | | | | +| 7 | 2 | 13-16 | 15 | 6 | 193-256 | 23 | 10 | 3073-4096 | | | | + +**5.5.4** 압축된 데이터 스트림은 압축된 헤더 데이터 바로 뒤에서 시작됩니다. 압축된 데이터 스트림은 다음과 같이 해석될 수 있습니다: + +``` +반복 + 입력 스트림에서 헤더를 읽는다. + + 저장된 블록이면 + 바이트 정렬될 때까지 비트를 건너뛴다. + count와 count의 1의 보수를 읽는다. + count 바이트의 데이터 블록을 복사한다. + 그렇지 않으면 + 블록 종료 코드가 전송될 때까지 반복 + 입력 스트림에서 리터럴 문자를 디코딩한다. + 리터럴 < 256이면 + 문자를 출력 스트림에 복사한다. + 그렇지 않으면 + 리터럴 = 블록 종료이면 + 반복을 중단한다. + 그렇지 않으면 + 입력 스트림에서 거리를 디코딩한다. + + 출력 스트림에서 distance 바이트만큼 뒤로 이동한 후, 이 + 위치에서 length개의 문자를 출력 스트림에 복사한다. + 반복 종료 +마지막 블록이 아닌 동안 반복 + +data descriptor가 존재하면 + 바이트 정렬될 때까지 비트를 건너뛴다. + crc와 크기를 읽는다. +끝 +``` + +### 5.6 Enhanced Deflating - 방법 9 + +**5.6.1** Enhanced Deflating 알고리즘은 Deflate와 유사하지만 최대 64K의 슬라이딩 딕셔너리를 사용합니다. Deflate64(tm)는 Deflate 추출기에서 지원됩니다. + +### 5.7 BZIP2 - 방법 12 + +**5.7.1** BZIP2는 Julian Seward가 개발한 오픈소스 데이터 압축 알고리즘입니다. 이 알고리즘에 대한 정보와 소스 코드는 인터넷에서 찾을 수 있습니다. + +### 5.8 LZMA - 방법 14 + +**5.8.1** LZMA는 Igor Pavlov가 개발하고 유지관리하는 블록 지향 범용 데이터 압축 알고리즘입니다. 마르코프 체인(Markov chain)과 범위 코더(range coder)를 활용하는 LZ77의 파생 알고리즘입니다. 이 알고리즘에 대한 정보와 소스 코드는 인터넷에서 찾을 수 있습니다. 사용에 관한 조건이나 제한 사항에 대해서는 이 알고리즘의 저자에게 문의하십시오. + +ZIP 형식 내 LZMA 지원은 다음과 같이 정의됩니다: + +**5.8.2** ZIP 로컬 헤더 및 중앙 헤더 레코드 내 Compression method 필드는 데이터가 LZMA를 사용하여 압축되었음을 나타내기 위해 값 14로 설정됩니다. + +**5.8.3** ZIP 로컬 헤더 및 중앙 헤더 레코드 내 Version needed to extract 필드는 이 기능을 지원하는 최소 ZIP 형식 버전을 나타내기 위해 6.3으로 설정됩니다. + +**5.8.4** LZMA 알고리즘을 사용하여 압축된 파일 데이터는 해당 파일의 로컬 헤더 바로 뒤에 반드시 위치해야 합니다(MUST). 표준 ZIP 암호화 헤더가 필요한 경우, 이는 로컬 헤더 뒤, LZMA 압축 파일 데이터 세그먼트 앞에 위치합니다. ZIP 형식 내 LZMA 압축 데이터 세그먼트의 위치는 다음과 같습니다: + +``` +[local header file 1] +[encryption header file 1] +[LZMA compressed data segment for file 1] +[data descriptor 1] +[local header file 2] +``` + +**5.8.5** 암호화 헤더와 데이터 디스크립터 레코드는 조건부로 존재할 수 있습니다(MAY). LZMA Compressed Data Segment는 다음과 같이 LZMA Properties Header 뒤에 LZMA Compressed Data가 이어지는 형태로 구성됩니다: + +``` +[LZMA properties header for file 1] +[LZMA compressed data for file 1] +``` + +**5.8.6** LZMA Compressed Data는 LZMA 압축 라이브러리가 제공하는 그대로 저장됩니다. 압축될 파일에 대한 압축 크기, 비압축 크기 및 그 밖의 파일 특성은 표준 ZIP 저장 형식으로 반드시 저장되어야 합니다(MUST). + +**5.8.7** LZMA Properties Header는 LZMA 압축 데이터를 압축 해제하는 데 필요한 특정 데이터를 저장합니다. 이 데이터는 LZMA SDK에 문서화된 WriteCoderProperties() 함수를 사용하여 LZMA 압축 엔진이 설정합니다. + +**5.8.8** LZMA Properties Header 내 속성 정보를 위한 저장 필드는 다음과 같습니다: + +``` +LZMA Version Information 2 bytes +LZMA Properties Size 2 bytes +LZMA Properties Data 가변, "LZMA Properties Size"로 정의됨 +``` + +**5.8.8.1** LZMA Version Information - 이 필드는 파일 압축에 사용된 LZMA SDK의 버전을 식별합니다. 첫 번째 바이트는 LZMA SDK의 주(major) 버전 번호를, 두 번째 바이트는 부(minor) 버전 번호를 저장합니다. + +**5.8.8.2** LZMA Properties Size - 이 필드는 나머지 속성 데이터의 크기를 정의합니다. 일반적으로 이 크기는 SDK의 버전에 따라 결정되어야 합니다(SHOULD). 이 크기 필드는 편의를 위해, 그리고 향후 이 압축 알고리즘의 변경으로 인해 발생할 수 있는 모호함을 방지하기 위해 포함되었습니다. + +**5.8.8.3** LZMA Property Data - 이 가변 크기 필드는 LZMA SDK에서 정의한, 압축 해제기에 필요한 값을 기록합니다. 이 필드에 저장되는 데이터는 "LZMA Version Information" 필드로 정의된 SDK 버전의 WriteCoderProperties()를 사용하여 얻어야 합니다(SHOULD). + +**5.8.8.4** "LZMA Properties Data" 필드의 레이아웃은 LZMA 압축 알고리즘의 함수입니다. 이 레이아웃은 시간이 지남에 따라 저자에 의해 변경될 수 있습니다(MAY). LZMA SDK 버전 4.3의 데이터 레이아웃은 딕셔너리 크기를 리틀 엔디안 순서로 저장하는 4바이트를 사용하는 5바이트 배열을 정의합니다. 이 배열의 첫 번째 요소로 다음 필드를 포함하는 단일 패킹된 바이트가 선행됩니다: + +``` +PosStateBits +LiteralPosStateBits +LiteralContextBits +``` + +이 필드들에 대한 더 자세한 설명은 LZMA 문서를 참고하십시오. + +**5.8.9** 방법 14(LZMA)로 압축된 데이터는 압축 데이터 스트림의 끝을 나타내는 EOS(end-of-stream) 마커를 포함할 수 있습니다(MAY). 이 마커는 필수는 아니지만, 처리를 용이하게 하기 위해 사용이 강력히 권장되며 구현자는 가능한 경우 항상 EOS 마커를 포함해야 합니다(SHOULD). EOS 마커를 사용하는 경우 범용 비트 1이 반드시 설정되어야 합니다(MUST). 범용 비트 1이 설정되지 않은 경우 EOS 마커는 존재하지 않습니다. + +### 5.9 WavPack - 방법 97 + +**5.9.1** 압축 방법 97의 사용에 대한 설명은 WinZIP International, LLC에서 제공한 것입니다. 이 방법은 David Bryant가 개발한 오픈소스 WavPack 오디오 압축 유틸리티에 의존합니다. WavPack에 대한 정보는 www.wavpack.com에서 확인할 수 있습니다. 사용에 관한 조건이나 제한 사항에 대해서는 이 알고리즘의 저자에게 문의하십시오. + +**5.9.2** 파일에 대한 WavPack 데이터는 로컬 헤더 데이터 끝 바로 뒤에서 시작됩니다. 이 데이터는 WavPack 압축 루틴의 출력입니다. ZIP 파일 내에서 WavPack 압축의 사용은 로컬 헤더와 중앙 디렉터리 헤더 모두에서 compression method 필드를 값 97로 설정하여 나타냅니다. Version needed to extract 및 version made by 필드는 Deflate 알고리즘으로 압축된 데이터에 사용되는 것과 동일한 값을 사용합니다. + +**5.9.3** ZIP 파일 내에서 WavPack 압축을 사용하여 디지털 샘플 데이터를 저장할 때의 구현 참고사항은, 샘플 데이터의 모든 바이트가 압축되어야 한다는 것입니다(SHOULD). 여기에는 바이트 경계까지의 사용되지 않는 비트도 포함됩니다. 예를 들어 12비트만 샘플 데이터에 사용하고 4비트가 사용되지 않는 2바이트 샘플이 있다고 합시다. WavPack 루틴에 샘플 크기로 12비트만 전달하면, 추출 시 원래 상태와 무관하게 사용되지 않는 4비트는 0으로 설정됩니다. 이를 방지하려면 샘플 데이터 크기의 전체 16비트를 제공해야 합니다(SHOULD). + +### 5.10 PPMd - 방법 98 + +**5.10.1** PPMd는 Dmitry Shkarin이 개발한 데이터 압축 알고리즘이며, Dmitry Subbotin이 개발한 carryless rangecoder를 포함합니다. 이 알고리즘은 여러 차수(order)의 컨텍스트에 대한 예측적 구문 일치(predictive phrase matching)를 기반으로 합니다. 이 알고리즘에 대한 정보와 소스 코드는 인터넷에서 찾을 수 있습니다. 사용에 관한 조건이나 제한 사항에 대해서는 이 알고리즘의 저자에게 문의하십시오. + +**5.10.2** 현재 ZIP 형식 내 PPMd 지원은 알고리즘의 버전 I, 리비전 1에 대해서만 제공됩니다. 이 알고리즘을 사용하기 위한 저장 요구사항은 다음과 같습니다. + +**5.10.3** 알고리즘 제어에 필요한 파라미터는 압축 데이터 바로 앞의 2바이트에 저장됩니다. 이 바이트들은 다음 필드를 저장하는 데 사용됩니다: + +``` +Model order - 최대 모델 차수를 설정, 기본값은 8, 가능한 값은 2에서 16까지 + +Sub-allocator size - sub-allocator 크기를 MB 단위로 설정, 기본값은 50, + 가능한 값은 1MB에서 256MB까지 + +Model restoration method - 메모리 부족 시 컨텍스트 모델을 재시작하는 데 + 사용되는 방법을 설정, 값은: + + 0 - 모델을 처음부터 다시 시작 - 기본값 + 1 - 모델을 잘라냄 - 성능을 최대 2배까지 저하시킴 + 2 - 컨텍스트 트리를 동결 - 권장하지 않음 +``` + +**5.10.4** 이 필드들을 2바이트 저장 필드에 패킹하는 예시는 다음과 같습니다. 이 값들은 인텔 low-byte/high-byte 순서로 저장됩니다. + +``` +wPPMd = (Model order - 1) + + ((Sub-allocator size - 1) << 4) + + (Model restoration method << 12) +``` + +### 5.11 AE-x Encryption marker - 방법 99 + +(APPENDIX E 참고) + +### 5.12 JPEG variant - 방법 96 + +### 5.13 PKWARE Data Compression Library Imploding - 방법 10 + +### 5.14 예약됨 - 방법 11 + +### 5.15 예약됨 - 방법 13 + +### 5.16 예약됨 - 방법 15 + +### 5.17 IBM z/OS CMPSC Compression - 방법 16 + +방법 16은 대부분의 IBM 메인프레임에서 사용 가능한 IBM 하드웨어 압축 기능을 활용합니다. 하드웨어 압축은 데이터 압축 속도를 크게 높일 수 있습니다. 이 방법은 LZ78 알고리즘의 변형을 사용합니다. CMPSC 하드웨어 압축은 COMPRESSION CALL 명령어를 사용하여 수행됩니다. + +ZIP 아카이브는 CP 명령어를 지원하는 메인프레임에서만 이 방법을 사용하여 생성할 수 있습니다. 추출은 이 압축 알고리즘을 지원하는 모든 플랫폼에서 이루어질 수 있습니다(MAY). 이 알고리즘을 사용하려면 압축 딕셔너리와 확장 딕셔너리를 생성해야 합니다. 확장 딕셔너리는 추출이 이루어질 시스템에서 사용할 수 있도록 ZIP 아카이브에 반드시 포함되어야 합니다(MUST). + +이 압축 알고리즘 및 딕셔너리에 대한 추가 정보는 IBM이 제공하는 문서 "IBM ESA/390 Data Compression"(SA22-7208-01)에서 찾을 수 있습니다. CMPSC 압축을 사용하기 위한 저장 요구사항은 다음과 같습니다. + +로컬 헤더 뒤에 ZIP 아카이브에 배치되는 압축 데이터 스트림의 형식은 다음과 같습니다: + +``` +[dictionary header] +[expansion dictionary] +[CMPSC compressed data] +``` + +CMPSC로 압축된 파일을 암호화에 사용하는 경우, 이 섹션들은 하나의 개체로서 반드시 암호화되어야 합니다(MUST). + +dictionary header의 형식은 다음과 같습니다: + +| 값 | 크기 | 설명 | +|---|---|---| +| Version | 1바이트 | 1 | +| Flags/Symsize | 1바이트 | 처리 플래그 및 심볼 크기 | +| DictionaryLen | 4바이트 | 확장 딕셔너리의 길이 | + +처리 플래그 및 심볼 크기 설명: + +상위 4비트는 처리 플래그를 저장하는 데 사용됩니다. 하위 4비트는 심볼의 크기를 비트 단위로 나타냅니다(값 범위는 9-13). 플래그 값은 아래에 정의되어 있습니다. + +``` +0x80 - 확장 딕셔너리 +0x40 - 확장 딕셔너리가 Deflate로 압축됨 +0x20 - 예약됨 +0x10 - 예약됨 +``` + +### 5.18 예약됨 - 방법 17 + +### 5.19 IBM TERSE - 방법 18 + +### 5.20 IBM LZ77 z Architecture - 방법 19 + +--- + +## 6.0 전통적인 PKWARE 암호화 (Traditional PKWARE Encryption) + +**6.0.1** 다음 정보는 전통적인 PKWARE 암호화를 지원하는 데 필요한 복호화 절차를 설명합니다. 이 형태의 암호화는 오늘날의 기준으로는 취약한 것으로 간주되며, 보안 요구가 낮은 상황이나 이전 .ZIP 애플리케이션과의 호환성을 위한 용도로만 사용을 권장합니다. + +### 6.1 전통적인 PKWARE 복호화 + +**6.1.1** PKWARE는 PKWARE의 전통적인 암호화 개발에 전문적으로 기여해 주신 Roger Schlafly 씨에게 감사드립니다. + +**6.1.2** PKZIP은 압축된 데이터 스트림을 암호화합니다. 암호화된 파일은 원래 형태로 추출되기 전에 반드시 복호화되어야 합니다(MUST). + +**6.1.3** 암호화된 각 파일은 해당 파일의 암호화 헤더를 정의하는 12바이트가 데이터 영역 시작 부분에 추가로 저장됩니다. 암호화 헤더는 원래 무작위 값으로 설정된 후, 세 개의 32비트 키를 사용하여 그 자체가 암호화됩니다. 키 값은 제공된 암호화 비밀번호를 사용하여 초기화됩니다. 각 바이트가 암호화된 후, 키는 본 문서 다른 곳에서 설명한 PKZIP에서 사용하는 것과 동일한 CRC-32 알고리즘과 결합된 의사난수 생성 기법을 사용하여 갱신됩니다. + +**6.1.4** 파일을 복호화하는 데 필요한 기본 단계는 다음과 같습니다: + +``` +1) 비밀번호로 세 개의 32비트 키를 초기화한다. +2) 12바이트 암호화 헤더를 읽고 복호화하여 암호화 키를 추가로 초기화한다. +3) 암호화 키를 사용하여 압축된 데이터 스트림을 읽고 복호화한다. +``` + +**6.1.5 암호화 키 초기화** + +``` +Key(0) <- 305419896 +Key(1) <- 591751049 +Key(2) <- 878082192 + +i <- 0부터 length(password)-1까지 반복 + update_keys(password(i)) +반복 종료 +``` + +update_keys()는 다음과 같이 정의됩니다: + +``` +update_keys(char): + Key(0) <- crc32(key(0),char) + Key(1) <- Key(1) + (Key(0) & 000000ffH) + Key(1) <- Key(1) * 134775813 + 1 + Key(2) <- crc32(key(2),key(1) >> 24) +end update_keys +``` + +crc32(old_crc,char)는 CRC 값과 문자가 주어지면, 본 문서 다른 곳에서 설명한 CRC-32 알고리즘을 적용한 갱신된 CRC 값을 반환하는 루틴입니다. + +**6.1.6 암호화 헤더 복호화** + +이 단계의 목적은 평문 공격(plaintext attack)을 무력화하기 위해 무작위 데이터를 기반으로 암호화 키를 추가로 초기화하는 것입니다. + +12바이트 암호화 헤더를 Buffer(0)부터 Buffer(11) 위치에 읽어 들입니다. + +``` +i <- 0부터 11까지 반복 + C <- buffer(i) ^ decrypt_byte() + update_keys(C) + buffer(i) <- C +반복 종료 +``` + +decrypt_byte()는 다음과 같이 정의됩니다: + +``` +unsigned char decrypt_byte() + local unsigned short temp + temp <- Key(2) | 2 + decrypt_byte <- (temp * (temp ^ 1)) >> 8 +end decrypt_byte +``` + +헤더가 복호화된 후, Buffer의 마지막 1바이트 또는 2바이트는 복호화 중인 파일의 CRC 상위 워드/바이트여야 하며(SHOULD), 인텔 low-byte/high-byte 순서로 저장됩니다. 2.0 이전 버전의 PKZIP은 2바이트 CRC 검사를 사용했으며, 2.0 이후 버전에서는 1바이트 CRC 검사를 사용합니다. 이를 통해 제공된 비밀번호가 올바른지 여부를 테스트할 수 있습니다. + +**6.1.7 압축된 데이터 스트림 복호화** + +압축된 데이터 스트림은 다음과 같이 복호화할 수 있습니다: + +``` +완료될 때까지 반복 + 문자를 읽어 C에 넣는다. + Temp <- C ^ decrypt_byte() + update_keys(temp) + Temp를 출력한다. +반복 종료 +``` + +--- + +## 7.0 강력 암호화 명세 (Strong Encryption Specification) + +**7.0.1** 본 명세서에 정의된 강력 암호화 기술의 일부는 특허 및 출원 중인 특허 신청의 적용을 받습니다. 자세한 정보는 본 문서의 "제품에 PKWARE 독점 기술 반영하기" 섹션을 참고하십시오. + +### 7.1 강력 암호화 개요 + +**7.1.1** 본 명세서의 버전 5.x는 강력 암호화 알고리즘에 대한 지원을 도입했습니다. 이 알고리즘들은 비밀번호 또는 X.509v3 디지털 인증서와 함께 사용하여 각 파일을 암호화할 수 있습니다. 본 형식 명세서는 오늘날의 보안 요구를 충족하고, PKI 환경과 비-PKI 환경 모두에서 사용자 간 상호운용성을 가능하게 하며, ZIP 프로그램을 실행하는 서로 다른 컴퓨팅 플랫폼 간의 상호운용성을 보장하기 위해 비밀번호 기반 또는 인증서 기반 암호화를 모두 지원합니다. + +**7.1.2** 비밀번호 기반 암호화는 사람들에게 가장 친숙한 형태의 암호화입니다. 그러나 비밀번호에 내재된 취약점(예: 사전/무차별 대입 공격에 대한 취약성)과 비밀번호 관리 및 지원 문제로 인해, 인증서 기반 암호화가 더 안전하고 확장 가능한 선택지가 됩니다. 업계의 노력과 지원은 전통적인 비밀번호 기반 암호화보다 확장성, 관리 옵션, 더 견고한 보안을 제공한다는 이유로 X.509v3 디지털 인증서와 공개 키 기반 구조(PKI)를 중심으로 한 더 발전된 보안 솔루션의 정의 및 도입으로 이동하고 있습니다. + +**7.1.3** 대부분의 표준 암호화 알고리즘이 본 명세서에서 지원됩니다. 이러한 알고리즘 다수에 대한 참조 구현은 상용 또는 오픈소스 배포자로부터 제공됩니다. 쉽게 구할 수 있는 암호화 툴킷을 통해 암호화 기능 구현이 간단해집니다. 본 문서는 데이터 암호화 원리나 이론에 대한 논문을 제공하려는 것이 아닙니다. 본 문서의 목적은 .ZIP 형식 내에서 상호운용 가능한 데이터 암호화를 구현하는 데 필요한 데이터 구조를 문서화하는 것입니다. 계속 읽기 전에 데이터 암호화에 대해 잘 이해하고 있을 것을 강력히 권장합니다. + +**7.1.4** 본 명세서 버전 5.0에서 도입된 알고리즘은 다음과 같습니다: + +``` +RC2 40비트, 64비트, 128비트 +RC4 40비트, 64비트, 128비트 +DES +3DES 112비트, 168비트 +``` + +버전 5.1은 다음에 대한 지원을 추가합니다: + +``` +AES 128비트, 192비트, 256비트 +``` + +**7.1.5** 버전 6.1은 OAEP 강화 표준을 지원하지 않는 스마트카드 및 USB 토큰 인증서 저장 방식과의 상호운용성을 지원하기 위한 암호화 데이터 변경을 도입합니다. + +**7.1.6** 버전 6.2는 중앙 디렉터리 데이터 구조를 압축 및 암호화하여 정보 유출을 줄임으로써 메타데이터를 암호화하는 지원을 도입합니다. 정보 유출은 파일이 암호화되어 저장되어 있음에도 불구하고, 레거시 ZIP 애플리케이션에서 해당 파일에 대한 정보가 노출됨으로써 발생할 수 있습니다. 노출되는 정보는 본 명세서에서 정의한 레코드와 필드에 저장된 파일 특성으로 구성되며, 파일명, 원본 크기, 타임스탬프, CRC32 값 등의 데이터가 포함됩니다. + +**7.1.7** 버전 6.3은 Blowfish 및 Twofish 알고리즘을 사용한 데이터 암호화 지원을 도입합니다. 이들은 Bruce Schneier가 개발한 대칭 블록 암호(symmetric block cipher)입니다. Blowfish는 32비트에서 448비트까지의 가변 길이 키 사용을 지원합니다. 블록 크기는 64비트입니다. 구현체는 16라운드를 사용해야 하며(SHOULD), ZIP 파일 내에서 지원되는 유일한 모드는 CBC입니다. Twofish는 128, 192, 256비트의 키 크기를 지원합니다. 블록 크기는 128비트입니다. 구현체는 16라운드를 사용해야 하며(SHOULD), ZIP 파일 내에서 지원되는 유일한 모드는 CBC입니다. Blowfish와 Twofish 알고리즘 모두에 대한 정보와 소스 코드는 인터넷에서 찾을 수 있습니다. 사용에 관한 조건이나 제한 사항에 대해서는 이 알고리즘의 저자에게 문의하십시오. + +**7.1.8** 중앙 디렉터리 암호화는 중앙 디렉터리 구조를 암호화하고, 암호화되지 않은 로컬 헤더에 중복 저장된 키 값을 마스킹함으로써 정보 유출에 대한 더 강력한 보호를 제공합니다. 암호화된 중앙 디렉터리 구조를 해석할 수 없는 ZIP 호환 프로그램은 압축 해제 정보를 위해 해당 로컬 헤더의 데이터를 신뢰할 수 없습니다. + +**7.1.9** 노출되어서는 안 되는 파일 정보를 포함할 수 있는 Extra Field 레코드는 로컬 헤더에 저장하지 않아야 하며(SHOULD NOT), 암호화될 수 있는 중앙 디렉터리에만 기록해야 합니다(SHOULD). 이 설계는 현재 스트리밍을 지원하지 않습니다. End of Central Directory record, Zip64 End of Central Directory Locator, Zip64 End of Central Directory record의 정보는 암호화되지 않습니다. 중앙 디렉터리가 암호화된 ZIP 파일 내 파일들에 대한 데이터를 보려면, 아카이브 내 파일이나 파일에 대한 정보를 보기 전에 복호화를 위한 적절한 비밀번호나 개인 키가 필요합니다. + +**7.1.10** 중앙 디렉터리 암호화 기능을 알지 못하는 이전 ZIP 호환 프로그램은 더 이상 중앙 디렉터리를 인식할 수 없으며, ZIP 파일이 손상되었다고 판단할 수 있습니다(MAY). 로컬 헤더를 사용하여 스트리밍 접근을 시도하는 프로그램은 각 파일에 대해 잘못된 정보를 보게 됩니다. 중앙 디렉터리 암호화가 모든 ZIP 파일에 사용될 필요는 없습니다. 더 강력한 보안을 위해 사용이 권장됩니다. 중앙 디렉터리 암호화를 사용하지 않는 ZIP 파일은 이전과 같이 동작해야 합니다(SHOULD). + +**7.1.11** 이 강력 암호화 기능 명세는 단순한 비밀번호 암호화부터 인증된 공개/개인 키 암호화에 이르기까지 확장 가능하고 플랫폼 간에 사용할 수 있는 암호화 요구를 지원하기 위한 것입니다. + +**7.1.12** 암호화는 데이터의 기밀성과 프라이버시를 제공합니다. 인증과 부인방지(non-repudiation)를 추가하기 위해 X.509 디지털 서명을 암호화와 결합할 것을 권장합니다. + +### 7.2 단일 비밀번호 대칭 암호화 방식 (Single Password Symmetric Encryption Method) + +**7.2.1** 강력 암호화 알고리즘을 사용하는 단일 비밀번호 대칭 암호화 방식은 본 형식에 정의된 전통적인 PKWARE 암호화와 유사하게 동작합니다. 강력 알고리즘의 처리 요구를 지원하기 위해 추가적인 데이터 구조가 추가됩니다. + +강력 암호화 데이터 구조는 다음과 같습니다: + +**7.2.2 General Purpose Bits** - 로컬 헤더와 중앙 헤더 레코드 모두의 범용 비트 플래그 중 비트 0과 6입니다. 두 비트가 모두 설정되면 강력 암호화를 나타냅니다. 비트 13은 설정된 경우 중앙 디렉터리가 암호화되었으며 로컬 헤더의 선택된 필드가 실제 값을 숨기도록 마스킹되었음을 나타냅니다. + +**7.2.3 Extra Field 0x0017 (중앙 헤더에만 존재)** + +이 레코드에서 고려할 필드는 다음과 같습니다: + +**7.2.3.1 Format** - 이 레코드의 데이터 형식 식별자입니다. 현재 허용되는 값은 정수값 2뿐입니다. + +**7.2.3.2 AlgId** - 다음 범위의 암호화 알고리즘 정수 식별자입니다: + +``` +0x6601 - DES +0x6602 - RC2 (version needed to extract < 5.2) +0x6603 - 3DES 168 +0x6609 - 3DES 112 +0x660E - AES 128 +0x660F - AES 192 +0x6610 - AES 256 +0x6702 - RC2 (version needed to extract >= 5.2) +0x6720 - Blowfish +0x6721 - Twofish +0x6801 - RC4 +0xFFFF - 알 수 없는 알고리즘 +``` + +**7.2.3.3 Bitlen** - 키의 명시적 비트 길이 (32 - 448비트) + +**7.2.3.4 Flags** - 복호화에 필요한 처리 플래그 + +``` +0x0001 - 복호화에 비밀번호 필요 +0x0002 - 인증서만 해당 +0x0003 - 복호화에 비밀번호 또는 인증서 필요 + +0x0003보다 큰 값은 인증서 처리를 위해 예약됨 +``` + +**7.2.4 압축된 파일 데이터 앞에 오는 Decryption header 레코드** + +**-Decryption Header:** + +| 값 | 크기 | 설명 | +|---|---|---| +| IVSize | 2바이트 | 초기화 벡터(IV)의 크기 | +| IVData | IVSize | 이 파일에 대한 초기화 벡터 | +| Size | 4바이트 | 나머지 decryption header 데이터의 크기 | +| Format | 2바이트 | 이 레코드의 형식 정의 | +| AlgID | 2바이트 | 암호화 알고리즘 식별자 | +| Bitlen | 2바이트 | 암호화 키의 비트 길이 | +| Flags | 2바이트 | 처리 플래그 | +| ErdSize | 2바이트 | Encrypted Random Data의 크기 | +| ErdData | ErdSize | Encrypted Random Data | +| Reserved1 | 4바이트 | 인증서 처리용 예약 데이터 | +| Reserved2 | (var) | 인증서 처리용 예약 데이터 | +| VSize | 2바이트 | 비밀번호 검증 데이터의 크기 | +| VData | VSize-4 | 비밀번호 검증 데이터 | +| VCRC32 | 4바이트 | 비밀번호 검증 데이터의 표준 ZIP CRC32 | + +**7.2.4.1 IVData** - IV의 크기는 알고리즘의 블록 크기와 일치해야 합니다(SHOULD). IVData는 완전히 무작위 데이터일 수 있습니다. 무작위로 생성된 데이터의 크기가 블록 크기와 일치하지 않으면, 필요에 따라 0으로 채우거나 잘라내야 합니다(SHOULD). IVSize가 0이면 IV = CRC32 + 비압축 파일 크기(64비트 리틀 엔디안 부호 없는 정수 값)입니다. + +**7.2.4.2 Format** - 이 레코드의 데이터 형식 식별자입니다. 현재 허용되는 값은 정수값 3뿐입니다. + +**7.2.4.3 AlgId** - 다음 범위의 암호화 알고리즘 정수 식별자입니다 (7.2.3.2와 동일한 목록). + +**7.2.4.4 Bitlen** - 키의 명시적 비트 길이 (32 - 448비트) + +**7.2.4.5 Flags** - 복호화에 필요한 처리 플래그 (7.2.3.4와 동일) + +**7.2.4.6 ErdData** - Encrypted random data는 각 파일을 암호화하기 위한 파일 세션 키를 생성하는 데 사용되는 무작위 데이터를 저장하는 데 사용됩니다. 키를 파생시키는 데 사용되는 해시 데이터를 계산하는 데 SHA1이 사용됩니다. 파일 세션 키는 사용자가 제공한 비밀번호로부터 생성된 마스터 세션 키에서 파생됩니다. decryption header의 Flags 필드에 값 0x4000이 포함된 경우, ErdData 필드는 반드시 3DES를 사용하여 복호화되어야 합니다(MUST). 값 0x4000이 설정되지 않은 경우, ErdData 필드는 반드시 AlgId를 사용하여 복호화되어야 합니다(MUST). + +**7.2.4.7 Reserved1** - 인증서 처리를 위해 예약되며, 값이 0이면 Reserved2 데이터는 없습니다. 이 데이터 구조에 대한 자세한 내용은 Certificate Processing Method 아래의 설명을 참고하십시오. + +**7.2.4.8 Reserved2** - 존재하는 경우, Reserved2 데이터 구조의 크기는 이 필드의 처음 4바이트를 건너뛰고 다음 2바이트를 나머지 크기로 사용하여 확인합니다. 이 데이터 구조에 대한 자세한 내용은 Certificate Processing Method 아래의 설명을 참고하십시오. + +**7.2.4.9 VSize** - 이 크기 값에는 항상 VCRC32 데이터의 4바이트가 포함되며 4바이트보다 큽니다. + +**7.2.4.10 VData** - 비밀번호 검증을 위한 무작위 데이터입니다. 이 데이터의 길이는 VSize이며 VSize는 암호화 블록 크기의 배수여야 합니다(MUST). VCRC32는 VData의 체크섬 값입니다. VData와 VCRC32는 암호화된 상태로 저장되며 파일에 대한 암호화된 데이터 스트림을 시작합니다. + +**7.2.5 유용한 팁** + +**7.2.5.1** 강력 암호화는 항상 압축 후 파일에 적용됩니다. 블록 지향 알고리즘은 모두 CBC(Cypher Block Chaining) 모드로 동작합니다. AES 암호화에 사용되는 블록 크기는 16입니다. 그 밖의 모든 블록 알고리즘은 블록 크기 8을 사용합니다. Windows XP SP1 및 그 이전의 모든 Windows 버전의 암호화 라이브러리에서 발견된 RC2 알고리즘 구현의 불일치를 반영하기 위해 RC2에 대해 두 개의 ID가 정의되어 있습니다. 길이가 0인 파일은 암호화하지 않을 것을 권장하지만, 프로그램은 ZIP 파일 내에서 이러한 파일이 발견되면 추출할 수 있도록 준비되어 있어야 합니다(SHOULD). + +**7.2.5.2** 암호화 프로세스의 의사코드는 다음과 같습니다: + +``` +Password = GetUserPassword() +MasterSessionKey = DeriveKey(SHA1(Password)) +RD = CryptographicStrengthRandomData() +각 파일에 대해 + IV = CryptographicStrengthRandomData() + VData = CryptographicStrengthRandomData() + VCRC32 = CRC32(VData) + FileSessionKey = DeriveKey(SHA1(IV + RD) + ErdData = Encrypt(RD,MasterSessionKey,IV) + Encrypt(VData + VCRC32 + FileData, FileSessionKey,IV) +완료 +``` + +**7.2.5.3** 함수 이름과 파라미터 요구사항은 선택한 암호화 툴킷에 따라 달라집니다. 각 알고리즘의 참조 구현을 지원하는 거의 모든 툴킷을 사용할 수 있습니다. RSA BSAFE(r), OpenSSL, Microsoft CryptoAPI 라이브러리 모두 잘 동작하는 것으로 알려져 있습니다. + +### 7.3 단일 비밀번호 - 중앙 디렉터리 암호화 + +**7.3.1** 중앙 디렉터리 암호화는 .ZIP 형식 내에서 중앙 디렉터리 구조를 암호화함으로써 이루어집니다. 이는 .ZIP 파일 처리에 가장 흔히 사용되는 메타데이터를 캡슐화합니다. 중복을 위해 각 파일의 로컬 헤더에도 추가 메타데이터가 저장됩니다. 중앙 디렉터리를 암호화하여 메타데이터를 숨기는 과정은 로컬 헤더 내 데이터를 보호하지 않습니다. 로컬 헤더에 노출된 메타데이터로 인한 정보 유출을 방지하기 위해, 파일에 대한 정보를 담고 있는 필드는 마스킹됩니다. + +**7.3.2 로컬 헤더** + +마스킹은 로컬 헤더 내 파일에 대한 필드의 실제 내용을 거짓 정보로 대체합니다. 마스킹된 경우, 로컬 헤더는 스트리밍 접근에 적합하지 않으며 손상된 아카이브의 데이터 복구 옵션도 줄어듭니다. 기밀 데이터를 포함할 수 있는 Extra Data 필드는 로컬 헤더에 저장하지 않아야 합니다(SHOULD NOT). Version needed to extract 필드에 설정되는 값은 중앙 디렉터리 암호화 여부와 무관하게 파일을 추출하는 데 필요한 올바른 값이어야 합니다(SHOULD). 중앙 디렉터리가 암호화될 때 마스킹 대상이 되는 로컬 헤더 내 필드는 다음과 같습니다: + +| 필드명 | 마스킹 값 | +|---|---| +| compression method | 0 | +| last mod file time | 0 | +| last mod file date | 0 | +| crc-32 | 0 | +| compressed size | 0 | +| uncompressed size | 0 | +| file name (가변 크기) | 1 - 0xFFFFFFFFFFFFFFFF 범위의 16진수 값을 문자열로 표현한 것이며, 그 크기는 file name length 필드에 설정됨 | + +마스킹된 파일명으로 할당되는 16진수 값은 단순히 첫 번째 파일부터 1로 시작하여 각 파일마다 순차적으로 증가하는 값입니다. ZIP 파일에 대한 수정은 각 파일에 대해 서로 다른 값이 저장되는 원인이 될 수 있습니다(MAY). 호환성을 위해 로컬 헤더의 파일명 필드는 비워두지 않아야 합니다(SHOULD NOT). 본 명세서 버전 6.2 기준으로 Compression Method와 Compressed Size 필드는 아직 마스킹되지 않습니다. ZIP64 형식에서 0xFFFF 또는 0xFFFFFFFF 값을 가진 필드는 마스킹하지 않아야 합니다(SHOULD NOT). + +**7.3.3 중앙 디렉터리 암호화** + +중앙 디렉터리 암호화에는 Central Directory Signature 데이터, Zip64 End of Central Directory record, Zip64 End of Central Directory Locator, End of Central Directory record의 암호화는 포함되지 않습니다. ZIP 파일 코멘트 데이터는 절대 암호화되지 않습니다. + +중앙 디렉터리를 암호화하기 전에 선택적으로 압축할 수 있습니다(MAY). 압축은 필수는 아니지만, 저장 효율을 위해 이 구조는 암호화 전에 압축될 것으로 가정합니다. 마찬가지로, 본 명세서는 중앙 디렉터리를 암호화하지 않고도 압축하는 것을 지원합니다. 이 기능의 초기 구현에서는 파일에 적용된 암호화 방법이 중앙 디렉터리에 적용된 암호화와 일치한다고 가정합니다. + +중앙 디렉터리 암호화는 파일 암호화와 유사한 방식으로 이루어집니다. 암호화된 데이터 앞에는 decryption header가 옵니다. 이 decryption header는 Archive Decryption Header라고 합니다. 이 레코드의 필드는 각 암호화된 파일 앞에 오는 decryption header와 동일합니다. Archive Decryption Header의 위치는 Zip64 End of Central Directory record의 Start of the Central Directory 필드 값으로 결정됩니다. 중앙 디렉터리가 암호화된 경우, Zip64 End of Central Directory record는 항상 존재합니다. + +본 명세서 6.2 버전부터 모든 버전에서 Zip64 End of Central Directory record의 레이아웃은 버전 2 형식을 따릅니다. 버전 2 형식은 다음과 같습니다: + +이 레코드의 버전 1 형식 내 선행하는 고정 크기 필드는 변경되지 않고 유지됩니다. 버전 1과 버전 2 모두 레코드 시그니처는 0x06064b50입니다. Offset of Start of Central Directory With Respect to the Starting Disk Number라는 필드의 마지막 바이트 바로 뒤에서 이 레코드의 버전 2를 정의하는 새 필드가 시작됩니다. + +**7.3.4 버전 2를 위한 새 필드** + +참고: 모든 필드는 인텔 low-byte/high-byte 순서로 저장됩니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| Compression Method | 2바이트 | 중앙 디렉터리를 압축하는 데 사용된 방법 | +| Compressed Size | 8바이트 | 압축된 데이터의 크기 | +| Original Size | 8바이트 | 원본 비압축 크기 | +| AlgId | 2바이트 | 암호화 알고리즘 ID | +| BitLen | 2바이트 | 암호화 키 길이 | +| Flags | 2바이트 | 암호화 플래그 | +| HashID | 2바이트 | 해시 알고리즘 식별자 | +| Hash Length | 2바이트 | 해시 데이터의 길이 | +| Hash Data | (가변) | 해시 데이터 | + +Compression Method는 중앙 헤더의 해당 필드와 동일한 범위의 값을 허용합니다. + +Compressed Size와 Original Size 값은 압축 또는 암호화된 Central Directory Signature의 데이터를 포함하지 않습니다. + +AlgId, BitLen, Flags 필드는 0x0017 레코드 내 해당 필드와 동일한 범위의 값을 허용합니다. + +Hash ID는 중앙 디렉터리 데이터를 해시하는 데 사용된 알고리즘을 식별합니다. 이 데이터는 반드시 해시될 필요는 없으며, 그 경우 HashID와 Hash Length 값은 모두 0이 됩니다. HashID의 가능한 값은 다음과 같습니다: + +| 값 | 알고리즘 | +|---|---| +| 0x0000 | 없음 | +| 0x0001 | CRC32 | +| 0x8003 | MD5 | +| 0x8004 | SHA1 | +| 0x8007 | RIPEMD160 | +| 0x800C | SHA256 | +| 0x800D | SHA384 | +| 0x800E | SHA512 | + +**7.3.5** 중앙 디렉터리 데이터가 서명된 경우, 서명을 위해 중앙 디렉터리를 해시하는 데 사용된 것과 동일한 해시 알고리즘을 사용해야 합니다(SHOULD). 이는 처리 효율을 위해 권장되지만, 서명 프로세스와 무관하게 위 알고리즘 중 어느 것을 사용해도 무방합니다. + +Hash Data는 중앙 디렉터리에 대한 해시 데이터를 포함합니다. 이 데이터의 길이는 사용된 알고리즘에 따라 달라집니다. + +Version Needed to Extract는 62로 설정해야 합니다(SHOULD). + +Total Number of Entries on the Current Disk 값은 0이 됩니다. 중앙 디렉터리를 암호화할 때 이 레코드는 더 이상 임의 접근(random access)을 지원하지 않습니다. + +**7.3.6** 중앙 디렉터리가 압축 및/또는 암호화된 경우, End of Central Directory record는 Total Number of Entries in the Central Directory 값으로 0xFFFFFFFF를 저장합니다. Total Number of Entries in the Central Directory on this Disk 필드에 저장되는 값은 0이 됩니다. 실제 값은 Zip64 End of Central Directory record의 해당 필드에 저장됩니다. + +**7.3.7** 중앙 디렉터리의 복호화 및 압축 해제는 파일을 복호화하고 압축 해제하는 것과 동일한 방식으로 수행됩니다. + +### 7.4 인증서 처리 방식 (Certificate Processing Method) + +ZIP 파일 암호화를 위한 인증서 처리 방식은 다음 추가 데이터 필드를 정의합니다: + +**7.4.1 인증서 플래그 값** + +중앙 디렉터리 Extra Field의 0x0017 필드와 압축된 파일 데이터 앞에 오는 Decryption header 레코드 모두의 Flags 필드에 나타날 수 있는 추가 처리 플래그는 다음과 같습니다: + +``` +0x0007 - 향후 사용을 위해 예약됨 +0x000F - 향후 사용을 위해 예약됨 +0x0100 - non-OAEP 키 래핑이 사용되었음을 나타냅니다. 이 필드가 + 설정된 경우 version needed to extract는 최소 61이어야 + 합니다(MUST). 이는 ErdData를 사용하여 마스터 세션 키를 + 생성할 때 OAEP 키 래핑이 사용되지 않음을 의미합니다. +0x4000 - ErdData는 반드시 3DES-168을 사용하여 복호화되어야 하며, + 그렇지 않으면 파일 내용 암호화에 사용된 것과 동일한 + 알고리즘을 사용합니다. +0x8000 - 향후 사용을 위해 예약됨 +``` + +**7.4.2 CertData - Extra Field 0x0017 레코드의 인증서 데이터 구조** + +0x0017 레코드의 CertData 필드로 정의된 Extra Field 섹션 내에 인증서 데이터를 저장하는 데 사용되는 데이터 구조는 다음과 같습니다: + +| 값 | 크기 | 설명 | +|---|---|---| +| RCount | 4바이트 | 수신자 수 | +| HashAlg | 2바이트 | 해시 알고리즘 식별자 | +| HSize | 2바이트 | 해시 크기 | +| SRList | (var) | 해시된 수신자 공개 키의 단순 목록 | + +- **RCount**: 암호화에 사용된 공개 키를 가진 의도된 수신자의 수를 정의합니다. SRList의 요소 수를 식별합니다. +- **HashAlg**: 암호화에 사용되는 각 공개 키의 공개 키 해시를 계산하는 데 사용된 해시 알고리즘을 정의합니다. 현재 이 필드는 SHA-1에 대한 다음 값만 지원합니다: `0x8004 - SHA1` +- **HSize**: 해시된 공개 키의 크기를 정의합니다. +- **SRList**: 의도된 각 수신자의 해시된 공개 키에 대한 가변 길이 목록입니다. 이 목록의 각 요소는 HSize 크기입니다. SRList의 전체 크기는 RCount * HSize로 결정됩니다. + +**7.4.3 Reserved1 - Certificate Decryption Header Reserved1 Data** + +| 값 | 크기 | 설명 | +|---|---|---| +| RCount | 4바이트 | 수신자 수 | + +- **RCount**: 암호화에 사용된 공개 키를 가진 의도된 수신자의 수를 정의합니다. 아래 정의된 REList 필드의 요소 수를 정의합니다. + +**7.4.4 Reserved2 - Certificate Decryption Header Reserved2 Data Structures** + +| 값 | 크기 | 설명 | +|---|---|---| +| HashAlg | 2바이트 | 해시 알고리즘 식별자 | +| HSize | 2바이트 | 해시 크기 | +| REList | (var) | 수신자 데이터 요소 목록 | + +- **HashAlg**: 암호화에 사용되는 각 공개 키의 공개 키 해시를 계산하는 데 사용된 해시 알고리즘을 정의합니다. 현재 이 필드는 SHA-1에 대한 다음 값만 지원합니다: `0x8004 - SHA1` +- **HSize**: REHData에 정의된 해시된 공개 키의 크기를 정의합니다. +- **REList**: 수신자 데이터의 가변 길이 목록입니다. 이 목록의 각 요소는 다음과 같은 Recipient Element 데이터 구조로 구성됩니다: + +**Recipient Element (REList) 데이터 구조:** + +| 값 | 크기 | 설명 | +|---|---|---| +| RESize | 2바이트 | REHData + REKData의 크기 | +| REHData | HSize | 수신자 공개 키의 해시 | +| REKData | (var) | 단순 키 블롭 | + +- **RESize**: 개별 REList 요소의 크기를 정의합니다. 이 값은 REHData 필드 + REKData 필드의 결합된 크기입니다. REHData는 HSize로 정의됩니다. REKData는 가변이며 RESize와 HSize를 사용하여 각 REList 요소에 대해 계산할 수 있습니다. +- **REHData**: 이 수신자에 대한 해시된 공개 키입니다. +- **REKData**: 단순 키 블롭입니다. 이 데이터 구조의 형식은 Microsoft CryptoAPI에서 정의되고 CryptExportKey() 함수를 사용하여 생성되는 것과 동일합니다. 현재 지원되는 단순 키 블롭의 버전은 Microsoft가 정의한 0x02입니다. + +### 7.5 인증서 처리 - 중앙 디렉터리 암호화 + +**7.5.1** 디지털 인증서를 사용한 중앙 디렉터리 암호화는 단일 비밀번호 중앙 디렉터리 암호화와 유사한 방식으로 동작합니다. 이 레코드는 그 안에 넣을 데이터가 있을 때만 존재합니다. 현재 이 레코드에는 ZIP 파일 내 파일을 암호화하거나 서명하는 데 디지털 인증서가 사용될 때 데이터가 채워집니다. 인증서 암호화나 디지털 서명 없이 비밀번호 암호화만 사용되는 경우, 현재로서는 이 레코드가 필요하지 않습니다. 존재하는 경우, 이 레코드는 실제 중앙 디렉터리 데이터 구조가 시작되기 전에 나타나며, 중앙 디렉터리가 암호화된 경우 Archive Decryption Header 바로 뒤에 위치합니다. + +**7.5.2** Archive Extra Data record는 다음 정보를 저장하는 데 사용됩니다. 향후 버전에서 추가 데이터가 추가될 수 있습니다(MAY). + +**Extra Data 필드:** + +``` +0x0014 - X.509 인증서용 PKCS#7 스토어 +0x0016 - 중앙 디렉터리에 대한 X.509 인증서 ID 및 서명 +0x0019 - PKCS#7 암호화 수신자 인증서 목록 +``` + +디지털 인증서 처리를 위해 원래는 중앙 디렉터리의 첫 번째 레코드에 위치했을 0x0014와 0x0016 Extra Data 레코드가 있습니다. 중앙 디렉터리를 암호화하거나 압축할 때, 0x0014와 0x0016 레코드는 반드시 Archive Extra Data record에 위치해야 하며(MUST), 첫 번째 중앙 디렉터리 레코드에 남아있어서는 안 됩니다(SHOULD NOT). Archive Extra Data record는 0x0019 데이터를 저장하는 데도 사용됩니다. + +**7.5.3** 존재하는 경우, Archive Extra Data record의 크기는 중앙 디렉터리의 크기에 포함됩니다. Archive Extra Data record의 데이터도 중앙 디렉터리 데이터 구조와 함께 압축 및 암호화됩니다. + +### 7.6 인증서 처리 방식의 차이점 + +**7.6.1** 인증서 처리 암호화 방식은 단일 비밀번호 대칭 암호화 방식과 다음과 같이 다릅니다. 사용자 정의 비밀번호를 사용하여 마스터 세션 키를 생성하는 대신, 암호학적으로 무작위인 데이터가 사용됩니다. 그런 다음 이 키 자료는 표준 키 래핑(key-wrapping) 기법을 사용하여 래핑됩니다. 이 키 자료는 자신의 개인 키를 사용하여 파일을 복호화해야 하는 각 수신자의 공개 키를 사용하여 래핑됩니다. + +**7.6.2** 본 명세서는 현재 디지털 인증서가 1024비트 이상의 RSA 형식 디지털 인증서에 대한 X.509 V3 형식을 따른다고 가정합니다. 이 인증서 처리 방식을 구현하려면 키 접근 및 관리를 위한 지원 로직이 필요합니다. 이 로직은 본 명세서의 범위를 벗어납니다. + +### 7.7 인증서 기반 암호화에서의 OAEP 처리 + +**7.7.1** OAEP는 Optimal Asymmetric Encryption Padding의 약자입니다. 복호화 키와 같은 작은 인코딩된 항목에 사용되는 강화 기법입니다. 이는 일반적으로 암호화 키 래핑 기법에 적용되며 PKCS #1에서 지원됩니다. 본 명세서의 버전 5.0과 6.0은 추가 보안을 위해 인증서 기반 복호화 키에 대한 OAEP 키 래핑을 지원하도록 설계되었습니다. + +**7.7.2** 스마트카드나 토큰에 저장된 개인 키에 대한 지원은 이 OAEP 로직과 충돌을 일으켰습니다. 대부분의 카드 및 토큰 제품은 OAEP 키 래핑된 데이터에 적용되는 추가 강화 기능을 지원하지 않습니다. 이 충돌을 해결하기 위해, 본 명세서의 버전 6.1 이상에서는 디지털 인증서를 사용한 암호화 시 더 이상 OAEP를 지원하지 않습니다. + +**7.7.3** 인증서 처리 방식의 초기 개발 시점에 제공된 PKZIP 버전은 파일의 version needed to extract 필드에 값 61을 설정했습니다. 이는 non-OAEP 키 래핑이 사용됨을 나타냅니다. 이는 인증서 암호화에만 영향을 미치며, 비밀번호 암호화 기능은 이 값의 영향을 받지 않아야 합니다(SHOULD NOT). 즉, 값 61은 인증서로만 암호화된 파일이나 비밀번호 암호화와 인증서 암호화를 모두 사용하여 암호화된 파일에서 발견될 수 있습니다(MAY). 두 방식 모두로 암호화된 파일은 문서화된 비밀번호 방식을 사용하여 안전하게 복호화할 수 있습니다. + +### 7.8 추가 암호화/복호화 데이터 레코드 + +**7.8.1** 위에서 정의한 강력한 비밀번호 및 인증서 암호화 방식을 지원하기 위해 ZIP 파일 내에 추가 정보가 저장될 수 있습니다(MAY). 여기에는 다음 레코드 유형이 포함되지만 이에 국한되지 않습니다: + +``` +0x0021 Policy Decryption Key Record +0x0022 Smartcrypt Key Provider Record +0x0023 Smartcrypt Policy Key Data Record +``` + +--- + +## 8.0 ZIP 파일 분할 및 스패닝 + +### 8.1 스패닝된(Spanned) ZIP 파일 + +**8.1.1** 스패닝은 ZIP 파일을 여러 이동식 매체에 걸쳐 나누는 과정입니다. 이 지원은 일반적으로 DOS 형식의 플로피 디스켓에 대해서만 제공되어 왔습니다. + +### 8.2 분할(Split) ZIP 파일 + +**8.2.1** 파일 분할은 스패닝에서 더 새롭게 파생된 방식입니다. 분할은 스패닝과 동일한 세그먼트화 과정을 따르지만, 각 세그먼트를 고유한 이동식 매체에 기록할 필요가 없으며, 대신 파일 시스템, 로컬 드라이브, 폴더 등 로컬 또는 비이동식 위치에 모든 조각을 배치하는 것을 지원합니다. + +### 8.3 파일 이름 지정 방식의 차이 + +**8.3.1** 스패닝된 ZIP 파일과 분할된 ZIP 파일 간의 주요 차이점은, 스패닝된 ZIP 파일의 모든 조각이 동일한 이름을 갖는다는 것입니다. 각 조각이 별도의 볼륨에 기록되므로 이름 충돌이 발생하지 않으며 각 세그먼트는 아카이브에 부여된 원래 .ZIP 파일 이름을 재사용할 수 있습니다. + +**8.3.2** DOS 스패닝 아카이브의 순서 지정은 DOS 볼륨 레이블을 사용하여 세그먼트 번호를 결정합니다. 각 세그먼트의 볼륨 레이블은 PKBACK#xxx 형식으로 기록되며, xxx는 001부터 nnn까지의 10진수 값으로 표현된 세그먼트 번호입니다. + +**8.3.3** 분할 ZIP 파일은 일반적으로 동일한 위치에 기록되며, 각 세그먼트가 동일한 드라이브에 위치하게 되므로 스패닝 이름 형식을 사용하면 이름 충돌이 발생할 수 있습니다. 이름 충돌을 방지하기 위해 분할 아카이브의 이름은 다음과 같이 지정됩니다. + +``` +세그먼트 1 = filename.z01 +세그먼트 n-1 = filename.z(n-1) +세그먼트 n = filename.zip +``` + +**8.3.4** 중앙 디렉터리를 빠르게 읽기 위해 마지막 세그먼트에는 .ZIP 확장자가 사용됩니다. 세그먼트 번호 n은 10진수 값이어야 합니다(SHOULD). + +### 8.4 스패닝된 자기 추출형 ZIP 파일 + +**8.4.1** 스패닝된 ZIP 파일은 PKSFX 자기 추출형 ZIP 파일일 수 있습니다(MAY). PKSFX 파일도 분할될 수 있지만(MAY), 이 경우 첫 번째 세그먼트는 반드시 filename.exe로 명명되어야 합니다(MUST). 분할된 PKSFX 아카이브의 첫 번째 세그먼트는 전체 실행 프로그램을 포함할 수 있을 만큼 충분히 커야 합니다(MUST). + +### 8.5 용량 및 마커 + +**8.5.1** 분할 아카이브의 용량은 다음과 같습니다: + +``` +최대 세그먼트 수 = 4,294,967,295 - 1 +최대 .ZIP 세그먼트 크기 = 4,294,967,295 바이트 +최소 세그먼트 크기 = 64K +최대 PKSFX 세그먼트 크기 = 2,147,483,647 바이트 +``` + +**8.5.2** 세그먼트 크기는 서로 다를 수 있지만(MAY), 관례적으로 마지막 세그먼트를 제외한 모든 세그먼트 크기는 동일해야 합니다(SHOULD)(마지막 세그먼트는 더 작을 수 있음, MAY). 로컬 및 중앙 디렉터리 헤더 레코드는 세그먼트 경계를 넘어 분할되어서는 안 됩니다(MUST NOT). 헤더 레코드를 기록할 때, 세그먼트 내 남은 바이트 수가 헤더 레코드의 크기보다 작으면 현재 세그먼트를 종료하고 다음 세그먼트의 시작 부분에 헤더를 기록합니다. 중앙 디렉터리는 세그먼트 경계에 걸쳐 있을 수 있지만(MAY), 중앙 디렉터리 내 어떤 단일 레코드도 세그먼트 간에 분할되어서는 안 됩니다(SHOULD NOT). + +**8.5.3** PKZIP for Windows(V2.50 이상), PKZIP Command Line(V2.50 이상), 또는 PKZIP Explorer로 생성된 스패닝/분할 아카이브는 아카이브 첫 번째 세그먼트의 처음 4바이트로 특수한 스패닝 시그니처를 포함합니다. 이 시그니처(0x08074b50) 바로 뒤에는 아카이브 내 첫 번째 파일의 로컬 헤더 시그니처가 이어집니다. + +**8.5.4** 스패닝이나 분할 과정이 시작되었지만 하나의 세그먼트만 필요한 경우에도 스패닝/분할 아카이브에 특수한 스패닝 마커가 나타날 수 있습니다(MAY). 이 경우 0x08074b50 시그니처는 임시 스패닝 마커 시그니처인 0x30304b50으로 대체됩니다. 분할 아카이브는 분할 아카이브를 생성하는 방법을 아는 다른 버전의 PKZIP으로만 압축 해제할 수 있습니다. + +**8.5.5** 시그니처 값 0x08074b50은 일부 ZIP 구현에서 Data Descriptor 레코드의 마커로도 사용됩니다. 이 대체 용도로 인한 충돌은 ZIP 파일 내 시그니처의 위치를 확인하여 의도된 용도를 판단함으로써 피할 수 있습니다. + +--- + +## 9.0 변경 절차 + +**9.1** .ZIP 파일 형식이 계속해서 유효한 기술로 남기 위해, 본 명세서는 주기적인 검토와 개정에 열려 있는 것으로 간주해야 합니다(SHOULD). 이 형식은 원래 어느 정도의 확장성을 염두에 두고 설계되었지만, 현재 또는 미래의 모든 기술 변화가 그 설계에서 반드시 고려된 것은 아니며 앞으로도 그럴 것입니다. + +**9.2** 애플리케이션에 본 형식의 확장 가능한 섹션에 대한 새로운 정의가 필요하거나, 새로운 데이터 구조나 새로운 기능을 제안하고자 하는 경우 zipformat@pkware.com으로 요청을 보내주십시오. 모든 제출물은 ZIP File Specification Committee가 검토하여 향후 버전의 본 명세서에 포함할지 여부를 결정합니다. + +**9.3** 본 명세서의 주기적인 개정판은 상호운용성을 보장하기 위해 DRAFT 또는 FINAL 상태로 발행됩니다. 명확성이나 내용 개선에 도움이 될 수 있는 의견과 피드백을 환영합니다. + +--- + +## 10.0 제품에 PKWARE 독점 기술 반영하기 + +**10.1** 강력 암호화 또는 패치와 관련된 APPNOTE 기술 구성 요소를 제품에 사용하거나 구현하려면 PKWARE와 별도로 체결한 실행 라이선스 계약이 필요합니다. 이러한 라이선스 취득에 관해서는 zipformat@pkware.com 또는 +1-414-289-9788로 PKWARE에 문의하십시오. + +**10.2** PKWARE 독점 기술에 대한 추가 정보는 http://www.pkware.com/appnote 에서 확인할 수 있습니다. + +--- + +## 11.0 감사의 말 + +PKZIP 및 PKUNZIP에 기여한 위에서 언급한 분들 외에도, PKWARE는 이 소프트웨어에 대해 .ZIP이라는 확장자를 제안해 주신 Robert Mahoney에게 특별히 감사드립니다. + +--- + +## 12.0 참고문헌 + +- Fiala, Edward R., and Greene, Daniel H., "Data compression with finite windows", Communications of the ACM, Volume 32, Number 4, April 1989, pages 490-505. +- Held, Gilbert, "Data Compression, Techniques and Applications, Hardware and Software Considerations", John Wiley & Sons, 1987. +- Huffman, D.A., "A method for the construction of minimum-redundancy codes", Proceedings of the IRE, Volume 40, Number 9, September 1952, pages 1098-1101. +- Nelson, Mark, "LZW Data Compression", Dr. Dobbs Journal, Volume 14, Number 10, October 1989, pages 29-37. +- Nelson, Mark, "The Data Compression Book", M&T Books, 1991. +- Storer, James A., "Data Compression, Methods and Theory", Computer Science Press, 1988. +- Welch, Terry, "A Technique for High-Performance Data Compression", IEEE Computer, Volume 17, Number 6, June 1984, pages 8-19. +- Ziv, J. and Lempel, A., "A universal algorithm for sequential data compression", Communications of the ACM, Volume 30, Number 6, June 1987, pages 520-540. +- Ziv, J. and Lempel, A., "Compression of individual sequences via variable-rate coding", IEEE Transactions on Information Theory, Volume 24, Number 5, September 1978, pages 530-536. + +--- + +## APPENDIX A - AS/400 Extra Field (0x0065) 속성 정의 + +### A.1 필드 정의 구조 + +``` +a. 길이를 포함한 필드 길이 2바이트, Big Endian +b. 필드 코드 2바이트 +c. 데이터 x바이트 +``` + +### A.2 필드 코드 설명 + +| 코드 | 설명 | +|---|---| +| 4001 | 소스 유형 (예: CLP 등) | +| 4002 | 라이브러리의 텍스트 설명 | +| 4003 | 파일의 텍스트 설명 | +| 4004 | 멤버의 텍스트 설명 | +| 4005 | x'F0' 또는 0은 PF-DTA, x'F1' 또는 1은 PF_SRC | +| 4007 | 데이터베이스 유형 코드 (1바이트) | +| 4008 | 데이터베이스 파일 및 필드 정의 | +| 4009 | GZIP 파일 유형 (2바이트) | +| 400B | IFS 코드 페이지 (2바이트) | +| 400C | IFS 마지막 파일 상태 변경 시간 (4바이트) | +| 400D | IFS 접근 시간 (4바이트) | +| 400E | IFS 수정 시간 (4바이트) | +| 005C | 파일 내 레코드 길이 (2바이트) | +| 0068 | GZIP 2 워드 (8바이트) | + +## APPENDIX B - z/OS Extra Field (0x0065) 속성 정의 + +### B.1 필드 정의 구조 + +``` +a. 길이를 포함한 필드 길이 2바이트, Big Endian +b. 필드 코드 2바이트 +c. 데이터 x바이트 +``` + +### B.2 필드 코드 설명 + +| 코드 | 설명 | +|---|---| +| 0001 | File Type (2바이트) | +| 0002 | NonVSAM Record Format (1바이트) | +| 0003 | 예약됨 | +| 0004 | NonVSAM Block Size (2바이트, Big Endian) | +| 0005 | Primary Space Allocation (3바이트, Big Endian) | +| 0006 | Secondary Space Allocation (3바이트, Big Endian) | +| 0007 | Space Allocation Type (1바이트 플래그) | +| 0008 | Modification Date (PKZIP 5.0+에서 지원 중단) | +| 0009 | Expiration Date (PKZIP 5.0+에서 지원 중단) | +| 000A | PDS Directory Block Allocation (3바이트, Big Endian 이진값) | +| 000B | NonVSAM Volume List (가변) | +| 000C | UNIT Reference (PKZIP 5.0+에서 지원 중단) | +| 000D | DF/SMS Management Class (8바이트 EBCDIC 텍스트) | +| 000E | DF/SMS Storage Class (8바이트 EBCDIC 텍스트) | +| 000F | DF/SMS Data Class (8바이트 EBCDIC 텍스트) | +| 0010 | PDS/PDSE Member Info. (30바이트) | +| 0011 | VSAM sub-filetype (2바이트) | +| 0012 | VSAM LRECL (13바이트 EBCDIC "(num_avg num_max)") | +| 0013 | VSAM Cluster Name (PKZIP 5.0+에서 지원 중단) | +| 0014 | VSAM KSDS Key Information (13바이트 EBCDIC "(num_length num_position)") | +| 0015 | VSAM Average LRECL (5바이트 EBCDIC, 공백 패딩) | +| 0016 | VSAM Maximum LRECL (5바이트 EBCDIC, 공백 패딩) | +| 0017 | VSAM KSDS Key Length (5바이트 EBCDIC, 공백 패딩) | +| 0018 | VSAM KSDS Key Position (5바이트 EBCDIC, 공백 패딩) | +| 0019 | VSAM Data Name (1-44바이트 EBCDIC 텍스트) | +| 001A | VSAM KSDS Index Name (1-44바이트 EBCDIC 텍스트) | +| 001B | VSAM Catalog Name (1-44바이트 EBCDIC 텍스트) | +| 001C | VSAM Data Space Type (9바이트 EBCDIC 텍스트) | +| 001D | VSAM Data Space Primary (9바이트 EBCDIC, 좌측 정렬) | +| 001E | VSAM Data Space Secondary (9바이트 EBCDIC, 좌측 정렬) | +| 001F | VSAM Data Volume List (가변, 6글자 볼륨 ID의 EBCDIC 텍스트 목록) | +| 0020 | VSAM Data Buffer Space (8바이트 EBCDIC, 좌측 정렬) | +| 0021 | VSAM Data CISIZE (5바이트 EBCDIC, 좌측 정렬) | +| 0022 | VSAM Erase Flag (1바이트 플래그) | +| 0023 | VSAM Free CI % (3바이트 EBCDIC, 좌측 정렬) | +| 0024 | VSAM Free CA % (3바이트 EBCDIC, 좌측 정렬) | +| 0025 | VSAM Index Volume List (가변, 6글자 볼륨 ID의 EBCDIC 텍스트 목록) | +| 0026 | VSAM Ordered Flag (1바이트 플래그) | +| 0027 | VSAM REUSE Flag (1바이트 플래그) | +| 0028 | VSAM SPANNED Flag (1바이트 플래그) | +| 0029 | VSAM Recovery Flag (1바이트 플래그) | +| 002A | VSAM WRITECHK Flag (1바이트 플래그) | +| 002B | VSAM Cluster/Data SHROPTS (3바이트 EBCDIC "n,y") | +| 002C | VSAM Index SHROPTS (3바이트 EBCDIC "n,y") | +| 002D | VSAM Index Space Type (9바이트 EBCDIC 텍스트) | +| 002E | VSAM Index Space Primary (9바이트 EBCDIC, 좌측 정렬) | +| 002F | VSAM Index Space Secondary (9바이트 EBCDIC, 좌측 정렬) | +| 0030 | VSAM Index CISIZE (5바이트 EBCDIC, 좌측 정렬) | +| 0031 | VSAM Index IMBED (1바이트 플래그) | +| 0032 | VSAM Index Ordered Flag (1바이트 플래그) | +| 0033 | VSAM REPLICATE Flag (1바이트 플래그) | +| 0034 | VSAM Index REUSE Flag (1바이트 플래그) | +| 0035 | VSAM Index WRITECHK Flag (1바이트 플래그, PKZIP 5.0+에서 지원 중단) | +| 0036 | VSAM Owner (8바이트 EBCDIC 텍스트) | +| 0037 | VSAM Index Owner (8바이트 EBCDIC 텍스트) | +| 0038-0057 | 예약됨 | +| 0058 | PDS/PDSE Member TTR Info. (6바이트, Big Endian) | +| 0059 | PDS 1st LMOD Text TTR (3바이트, Big Endian) | +| 005A | PDS LMOD EP Rec # (4바이트, Big Endian) | +| 005B | 예약됨 | +| 005C | Max Length of records (2바이트, Big Endian) | +| 005D | PDSE Flag (1바이트 플래그) | +| 005E-0064 | 예약됨 | +| 0065 | Last Date Referenced (4바이트, Packed Hex "yyyymmdd") | +| 0066 | Date Created (4바이트, Packed Hex "yyyymmdd") | +| 0068 | GZIP two words (8바이트) | +| 0071 | Extended NOTE Location (12바이트, Big Endian) | +| 0072 | Archive device UNIT (6바이트, EBCDIC) | +| 0073 | Archive 1st Volume (6바이트, EBCDIC) | +| 0074 | Archive 1st VOL File Seq# (2바이트, Binary) | +| 0075 | Native I/O Flags (2바이트) | +| 0081 | Unix File Type (1바이트, 열거형) | +| 0082 | Unix File Format (1바이트, 열거형) | +| 0083 | Unix File Character Set Tag Info (4바이트) | +| 0090 | ZIP Environmental Processing Info (4바이트) | +| 0091 | EAV EATTR Flags (1바이트) | +| 0092 | DSNTYPE Flags (1바이트) | +| 0093 | Total Space Allocation (Cyls) (4바이트, Big Endian) | +| 009D | NONVSAM DSORG (2바이트) | +| 009E | Program Virtual Object Info (3바이트) | +| 009F | Encapsulated file Info (9바이트) | +| 400C | Unix File Creation Time (4바이트) | +| 400D | Unix File Access Time (4바이트) | +| 400E | Unix File Modification time (4바이트) | +| 4101 | IBMCMPSC Compression Info (가변) | +| 4102 | IBMCMPSC Compression Size (8바이트, Big Endian) | + +## APPENDIX C - Zip64 Extensible Data Sector 매핑 + +**-Z390 Extra Field:** + +다음은 확장 테이프 작업을 위한 ZIP64 "extra" 블록 속성의 일반적인 레이아웃입니다. + +참고: 일부 필드는 빅 엔디안 형식으로 저장됩니다. 별도로 명시하지 않는 한 모든 텍스트는 EBCDIC 형식입니다. + +| 값 | 크기 | 설명 | +|---|---|---| +| (Z390) 0x0065 | 2바이트 | 이 "extra" 블록 유형의 태그 | +| Size | 4바이트 | 뒤따르는 데이터 블록의 크기 | +| Tag | 4바이트 | EBCDIC "Z390" | +| Length71 | 2바이트 | Big Endian | +| Subcode71 | 2바이트 | Enote 유형 코드 | +| FMEPos | 1바이트 | | +| Length72 | 2바이트 | Big Endian | +| Subcode72 | 2바이트 | Unit 유형 코드 | +| Unit | 1바이트 | Unit | +| Length73 | 2바이트 | Big Endian | +| Subcode73 | 2바이트 | Volume1 유형 코드 | +| FirstVol | 1바이트 | Volume | +| Length74 | 2바이트 | Big Endian | +| Subcode74 | 2바이트 | FirstVol 파일 시퀀스 | +| FileSeq | 2바이트 | 시퀀스 | + +## APPENDIX D - 언어 인코딩 (EFS) + +**D.1** ZIP 형식은 역사적으로 원본 IBM PC 문자 인코딩 세트(흔히 IBM Code Page 437로 불림)만 지원해 왔습니다. 이는 파일명 문자를 원본 MS-DOS 값 범위 내로만 저장하도록 제한하며, 다른 문자 인코딩이나 언어로 된 파일명을 제대로 지원하지 않습니다. 이러한 제한을 해결하기 위해 본 명세서는 다음 변경 사항을 지원합니다. + +**D.2** 범용 비트 11이 설정되지 않은 경우, 파일명과 코멘트는 원본 ZIP 문자 인코딩을 따라야 합니다(SHOULD). 범용 비트 11이 설정된 경우, 파일명과 코멘트는 UTF-8 저장 명세로 정의된 문자 인코딩 형식을 사용하여 Unicode Standard 버전 4.1.0 이상을 반드시 지원해야 합니다(MUST). Unicode Standard는 The Unicode Consortium(www.unicode.org)에서 발행합니다. ZIP 파일 내에 저장된 UTF-8 인코딩 데이터에는 바이트 순서 표시(BOM)가 포함되지 않을 것으로 예상됩니다. + +**D.3** 애플리케이션은 0x0008 Extra Field를 사용하여 이 파일명 저장 방식을 보완하도록 선택할 수 있습니다(MAY). 이 선택적 필드에 대한 저장 방식은 현재 정의되지 않았지만, 파일명이나 파일 내용 인코딩 작업을 추가로 지원할 수 있는 원본 또는 대상 인코딩에 대한 확장 정보를 저장하는 데 사용될 예정입니다. 이 필드를 어떻게 사용해야 하는지에 대한 요구사항이 있으면 PKWARE에 문의하십시오. + +**D.4** 0x0008 Extra Field 저장은 범용 비트 11의 설정값과 무관하게 사용될 수 있습니다(MAY). 이 필드의 의도된 사용 예로는 "modified-UTF-8"(JAVA)이 사용되는지 또는 UTF-8-MAC이 사용되는지를 저장하는 것이 있습니다. 마찬가지로 그 밖에 흔히 사용되는 문자 인코딩(코드 페이지) 지정도 이 필드를 통해 나타낼 수 있습니다. 0x0008 레코드 사용을 위한 공식화된 값은 현재 정의되지 않았습니다. 0x0008 필드의 레이아웃에 대한 정의는 확정되는 대로 발표될 예정입니다. 0x0008 Extra Field를 사용하면 IBM Code Page 437이나 UTF-8이 아닌 다른 인코딩으로 ZIP 파일 내에 데이터를 저장할 수 있습니다. + +**D.5** 범용 비트 11은 파일 내용이나 비밀번호의 인코딩을 암시하지 않습니다. 파일 내용이나 비밀번호에 대한 문자 인코딩을 정의하는 값은 반드시 0x0008 Extended Language Encoding Extra Field에 저장되어야 합니다(MUST). + +**D.6** Info-ZIP 그룹의 Ed Gordon은 UTF-8 파일명 및 파일 코멘트 필드를 저장하는 데 사용할 수 있는 한 쌍의 "extra field" 레코드를 정의했습니다. 이 레코드들은 표준 파일명 및 코멘트 필드에 UTF-8 데이터를 저장하는 범용 비트 11 방식이 바람직하지 않은 경우에 사용할 수 있습니다. 이 대체 방식이 흔히 사용되는 경우는 이전 프로그램과의 하위 호환성이 필요한 경우입니다. + +**D.7** 이 필드들의 레코드 구조에 대한 정의는 위의 "extra field" 레코드에 대한 서드파티 매핑 섹션에 포함되어 있습니다. 이 레코드들은 Header ID 0x6375(Info-ZIP Unicode Comment Extra Field)와 0x7075(Info-ZIP Unicode Path Extra Field)로 식별됩니다. + +**D.8** ZIP 파일을 작성할 때 어떤 저장 방식을 사용할지는 구현에 맡겨져 있습니다. 개발자는 ZIP 파일이 두 방식 중 어느 것이든 포함할 수 있다고 예상해야 하며(SHOULD), 두 형식 모두를 읽을 수 있도록 지원해야 합니다(SHOULD). 범용 비트 11을 사용하면 각 파일마다 추가 "extra field" 데이터가 필요하지 않으므로 파일명 데이터의 저장 요구가 줄어들지만, 이전 ZIP 프로그램이 파일을 추출하지 못하는 결과를 초래할 수 있습니다. 0x6375 및 0x7075 레코드를 사용하면 이전 ZIP 프로그램에서도 항상 읽을 수 있는 ZIP 파일을 만들 수 있지만(SHOULD), 파일명 및/또는 파일 코멘트 필드를 기록하는 데 파일당 더 많은 저장 공간이 필요합니다. + +## APPENDIX E - AE-x 암호화 마커 + +**E.1** AE-x는 Dr. Brian Gladman이 개발한 파일 암호화 유틸리티를 기반으로 하는, ZIP 파일에서 사용되는 대체 비밀번호 기반 암호화 방식을 정의합니다. Dr. Gladman의 방식에 대한 정보는 다음에서 확인할 수 있습니다: + +``` +http://www.gladman.me.uk/cryptography_technology/fileencrypt/ +``` + +**E.2** AE-x는 CTR(카운터 모드)과 HMAC-SHA1을 사용하는 AES를 사용합니다. 128비트 또는 256비트 키 크기를 사용하는 암호화를 정의합니다. 192비트 복호화에 대한 지원을 제한하지는 않습니다. + +**E.3** 이 방식은 파일이 암호화되었음을 나타내기 위해 범용 비트 플래그(섹션 4.4.4)의 표준 ZIP 암호화 비트(비트 0)를 사용합니다. + +**E.4** compression method 필드(섹션 4.4.5)는 파일이 이 방식을 사용하여 암호화되었음을 나타내기 위해 99로 설정됩니다. + +**E.5** 실제 압축 방법은 Header ID 0x9901로 식별되는 extra field 구조에 저장됩니다. 이 레코드 구조에 대한 정보는 http://www.winzip.com/aes_info.htm 에서 확인할 수 있습니다. + +**E.6** 0x9901 구조에 대해 두 가지 버전이 정의되어 있습니다. + +**E.6.1** 버전 1은 CRC-32 필드(섹션 4.4.7)에 파일 CRC 값을 저장합니다. + +**E.6.2** 버전 2는 CRC-32 필드에 값 0을 저장합니다. diff --git a/Docs/APPNOTE-6.39.9.txt b/Docs/APPNOTE-6.39.9.txt new file mode 100644 index 0000000..e07a35c --- /dev/null +++ b/Docs/APPNOTE-6.39.9.txt @@ -0,0 +1,3778 @@ +File: APPNOTE.TXT - .ZIP File Format Specification +Version: 6.3.9 +Status: FINAL - replaces version 6.3.8 +Revised: July 15, 2020 +Copyright (c) 1989 - 2014, 2018, 2019, 2020 PKWARE Inc., All Rights Reserved. + +1.0 Introduction +--------------- + +1.1 Purpose +----------- + + 1.1.1 This specification is intended to define a cross-platform, + interoperable file storage and transfer format. Since its + first publication in 1989, PKWARE, Inc. ("PKWARE") has remained + committed to ensuring the interoperability of the .ZIP file + format through periodic publication and maintenance of this + specification. We trust that all .ZIP compatible vendors and + application developers that use and benefit from this format + will share and support this commitment to interoperability. + +1.2 Scope +--------- + + 1.2.1 ZIP is one of the most widely used compressed file formats. It is + universally used to aggregate, compress, and encrypt files into a single + interoperable container. No specific use or application need is + defined by this format and no specific implementation guidance is + provided. This document provides details on the storage format for + creating ZIP files. Information is provided on the records and + fields that describe what a ZIP file is. + +1.3 Trademarks +-------------- + + 1.3.1 PKWARE, PKZIP, Smartcrypt, SecureZIP, and PKSFX are registered + trademarks of PKWARE, Inc. in the United States and elsewhere. + PKPatchMaker, Deflate64, and ZIP64 are trademarks of PKWARE, Inc. + Other marks referenced within this document appear for identification + purposes only and are the property of their respective owners. + + +1.4 Permitted Use +----------------- + + 1.4.1 This document, "APPNOTE.TXT - .ZIP File Format Specification" is the + exclusive property of PKWARE. Use of the information contained in this + document is permitted solely for the purpose of creating products, + programs and processes that read and write files in the ZIP format + subject to the terms and conditions herein. + + 1.4.2 Use of the content of this document within other publications is + permitted only through reference to this document. Any reproduction + or distribution of this document in whole or in part without prior + written permission from PKWARE is strictly prohibited. + + 1.4.3 Certain technological components provided in this document are the + patented proprietary technology of PKWARE and as such require a + separate, executed license agreement from PKWARE. Applicable + components are marked with the following, or similar, statement: + 'Refer to the section in this document entitled "Incorporating + PKWARE Proprietary Technology into Your Product" for more information'. + +1.5 Contacting PKWARE +--------------------- + + 1.5.1 If you have questions on this format, its use, or licensing, or if you + wish to report defects, request changes or additions, please contact: + + PKWARE, Inc. + 201 E. Pittsburgh Avenue, Suite 400 + Milwaukee, WI 53204 + +1-414-289-9788 + +1-414-289-9789 FAX + zipformat@pkware.com + + 1.5.2 Information about this format and a reference copy of this document + is publicly available at: + + http://www.pkware.com/appnote + +1.6 Disclaimer +-------------- + + 1.6.1 Although PKWARE will attempt to supply current and accurate + information relating to its file formats, algorithms, and the + subject programs, the possibility of error or omission cannot + be eliminated. PKWARE therefore expressly disclaims any warranty + that the information contained in the associated materials relating + to the subject programs and/or the format of the files created or + accessed by the subject programs and/or the algorithms used by + the subject programs, or any other matter, is current, correct or + accurate as delivered. Any risk of damage due to any possible + inaccurate information is assumed by the user of the information. + Furthermore, the information relating to the subject programs + and/or the file formats created or accessed by the subject + programs and/or the algorithms used by the subject programs is + subject to change without notice. + +2.0 Revisions +-------------- + +2.1 Document Status +-------------------- + + 2.1.1 If the STATUS of this file is marked as DRAFT, the content + defines proposed revisions to this specification which may consist + of changes to the ZIP format itself, or that may consist of other + content changes to this document. Versions of this document and + the format in DRAFT form may be subject to modification prior to + publication STATUS of FINAL. DRAFT versions are published periodically + to provide notification to the ZIP community of pending changes and to + provide opportunity for review and comment. + + 2.1.2 Versions of this document having a STATUS of FINAL are + considered to be in the final form for that version of the document + and are not subject to further change until a new, higher version + numbered document is published. Newer versions of this format + specification are intended to remain interoperable with all prior + versions whenever technically possible. + +2.2 Change Log +-------------- + + Version Change Description Date + ------- ------------------ ---------- + 5.2 -Single Password Symmetric Encryption 07/16/2003 + storage + + 6.1.0 -Smartcard compatibility 01/20/2004 + -Documentation on certificate storage + + 6.2.0 -Introduction of Central Directory 04/26/2004 + Encryption for encrypting metadata + -Added OS X to Version Made By values + + 6.2.1 -Added Extra Field placeholder for 04/01/2005 + POSZIP using ID 0x4690 + + -Clarified size field on + "zip64 end of central directory record" + + 6.2.2 -Documented Final Feature Specification 01/06/2006 + for Strong Encryption + + -Clarifications and typographical + corrections + + 6.3.0 -Added tape positioning storage 09/29/2006 + parameters + + -Expanded list of supported hash algorithms + + -Expanded list of supported compression + algorithms + + -Expanded list of supported encryption + algorithms + + -Added option for Unicode filename + storage + + -Clarifications for consistent use + of Data Descriptor records + + -Added additional "Extra Field" + definitions + + 6.3.1 -Corrected standard hash values for 04/11/2007 + SHA-256/384/512 + + 6.3.2 -Added compression method 97 09/28/2007 + + -Documented InfoZIP "Extra Field" + values for UTF-8 file name and + file comment storage + + 6.3.3 -Formatting changes to support 09/01/2012 + easier referencing of this APPNOTE + from other documents and standards + + 6.3.4 -Address change 10/01/2014 + + 6.3.5 -Documented compression methods 16 11/31/2018 + and 99 (4.4.5, 4.6.1, 5.11, 5.17, + APPENDIX E) + + -Corrected several typographical + errors (2.1.2, 3.2, 4.1.1, 10.2) + + -Marked legacy algorithms as no + longer suitable for use (4.4.5.1) + + -Added clarity on MS DOS time format + (4.4.6) + + -Assign extrafield ID for Timestamps + (4.5.2) + + -Field code description correction (A.2) + + -More consistent use of MAY/SHOULD/MUST + + -Expanded 0x0065 record attribute codes (B.2) + + -Initial information on 0x0022 Extra Data + + 6.3.6 -Corrected typographical error 04/26/2019 + (4.4.1.3) + + 6.3.7 -Added Zstandard compression method ID + (4.4.5) + + -Corrected several reported typos + + -Marked intended use for general purpose bit 14 + + -Added Data Stream Alignment Extra Data info + (4.6.11) + + 6.3.8 -Resolved Zstandard compression method ID conflict + (4.4.5) + + -Added additional compression method ID values in use + + 6.3.9 -Corrected a typo in Data Stream Alignment description + (4.6.11) + + + + +3.0 Notations +------------- + + 3.1 Use of the term MUST or SHALL indicates a required element. + + 3.2 MUST NOT or SHALL NOT indicates an element is prohibited from use. + + 3.3 SHOULD indicates a RECOMMENDED element. + + 3.4 SHOULD NOT indicates an element NOT RECOMMENDED for use. + + 3.5 MAY indicates an OPTIONAL element. + + +4.0 ZIP Files +------------- + +4.1 What is a ZIP file +---------------------- + + 4.1.1 ZIP files MAY be identified by the standard .ZIP file extension + although use of a file extension is not required. Use of the + extension .ZIPX is also recognized and MAY be used for ZIP files. + Other common file extensions using the ZIP format include .JAR, .WAR, + .DOCX, .XLSX, .PPTX, .ODT, .ODS, .ODP and others. Programs reading or + writing ZIP files SHOULD rely on internal record signatures described + in this document to identify files in this format. + + 4.1.2 ZIP files SHOULD contain at least one file and MAY contain + multiple files. + + 4.1.3 Data compression MAY be used to reduce the size of files + placed into a ZIP file, but is not required. This format supports the + use of multiple data compression algorithms. When compression is used, + one of the documented compression algorithms MUST be used. Implementors + are advised to experiment with their data to determine which of the + available algorithms provides the best compression for their needs. + Compression method 8 (Deflate) is the method used by default by most + ZIP compatible application programs. + + + 4.1.4 Data encryption MAY be used to protect files within a ZIP file. + Keying methods supported for encryption within this format include + passwords and public/private keys. Either MAY be used individually + or in combination. Encryption MAY be applied to individual files. + Additional security MAY be used through the encryption of ZIP file + metadata stored within the Central Directory. See the section on the + Strong Encryption Specification for information. Refer to the section + in this document entitled "Incorporating PKWARE Proprietary Technology + into Your Product" for more information. + + 4.1.5 Data integrity MUST be provided for each file using CRC32. + + 4.1.6 Additional data integrity MAY be included through the use of + digital signatures. Individual files MAY be signed with one or more + digital signatures. The Central Directory, if signed, MUST use a + single signature. + + 4.1.7 Files MAY be placed within a ZIP file uncompressed or stored. + The term "stored" as used in the context of this document means the file + is copied into the ZIP file uncompressed. + + 4.1.8 Each data file placed into a ZIP file MAY be compressed, stored, + encrypted or digitally signed independent of how other data files in the + same ZIP file are archived. + + 4.1.9 ZIP files MAY be streamed, split into segments (on fixed or on + removable media) or "self-extracting". Self-extracting ZIP + files MUST include extraction code for a target platform within + the ZIP file. + + 4.1.10 Extensibility is provided for platform or application specific + needs through extra data fields that MAY be defined for custom + purposes. Extra data definitions MUST NOT conflict with existing + documented record definitions. + + 4.1.11 Common uses for ZIP MAY also include the use of manifest files. + Manifest files store application specific information within a file stored + within the ZIP file. This manifest file SHOULD be the first file in the + ZIP file. This specification does not provide any information or guidance on + the use of manifest files within ZIP files. Refer to the application developer + for information on using manifest files and for any additional profile + information on using ZIP within an application. + + 4.1.12 ZIP files MAY be placed within other ZIP files. + +4.2 ZIP Metadata +---------------- + + 4.2.1 ZIP files are identified by metadata consisting of defined record types + containing the storage information necessary for maintaining the files + placed into a ZIP file. Each record type MUST be identified using a header + signature that identifies the record type. Signature values begin with the + two byte constant marker of 0x4b50, representing the characters "PK". + + +4.3 General Format of a .ZIP file +--------------------------------- + + 4.3.1 A ZIP file MUST contain an "end of central directory record". A ZIP + file containing only an "end of central directory record" is considered an + empty ZIP file. Files MAY be added or replaced within a ZIP file, or deleted. + A ZIP file MUST have only one "end of central directory record". Other + records defined in this specification MAY be used as needed to support + storage requirements for individual ZIP files. + + 4.3.2 Each file placed into a ZIP file MUST be preceded by a "local + file header" record for that file. Each "local file header" MUST be + accompanied by a corresponding "central directory header" record within + the central directory section of the ZIP file. + + 4.3.3 Files MAY be stored in arbitrary order within a ZIP file. A ZIP + file MAY span multiple volumes or it MAY be split into user-defined + segment sizes. All values MUST be stored in little-endian byte order unless + otherwise specified in this document for a specific data element. + + 4.3.4 Compression MUST NOT be applied to a "local file header", an "encryption + header", or an "end of central directory record". Individual "central + directory records" MUST NOT be compressed, but the aggregate of all central + directory records MAY be compressed. + + 4.3.5 File data MAY be followed by a "data descriptor" for the file. Data + descriptors are used to facilitate ZIP file streaming. + + + 4.3.6 Overall .ZIP file format: + + [local file header 1] + [encryption header 1] + [file data 1] + [data descriptor 1] + . + . + . + [local file header n] + [encryption header n] + [file data n] + [data descriptor n] + [archive decryption header] + [archive extra data record] + [central directory header 1] + . + . + . + [central directory header n] + [zip64 end of central directory record] + [zip64 end of central directory locator] + [end of central directory record] + + + 4.3.7 Local file header: + + local file header signature 4 bytes (0x04034b50) + version needed to extract 2 bytes + general purpose bit flag 2 bytes + compression method 2 bytes + last mod file time 2 bytes + last mod file date 2 bytes + crc-32 4 bytes + compressed size 4 bytes + uncompressed size 4 bytes + file name length 2 bytes + extra field length 2 bytes + + file name (variable size) + extra field (variable size) + + 4.3.8 File data + + Immediately following the local header for a file + SHOULD be placed the compressed or stored data for the file. + If the file is encrypted, the encryption header for the file + SHOULD be placed after the local header and before the file + data. The series of [local file header][encryption header] + [file data][data descriptor] repeats for each file in the + .ZIP archive. + + Zero-byte files, directories, and other file types that + contain no content MUST NOT include file data. + + 4.3.9 Data descriptor: + + crc-32 4 bytes + compressed size 4 bytes + uncompressed size 4 bytes + + 4.3.9.1 This descriptor MUST exist if bit 3 of the general + purpose bit flag is set (see below). It is byte aligned + and immediately follows the last byte of compressed data. + This descriptor SHOULD be used only when it was not possible to + seek in the output .ZIP file, e.g., when the output .ZIP file + was standard output or a non-seekable device. For ZIP64(tm) format + archives, the compressed and uncompressed sizes are 8 bytes each. + + 4.3.9.2 When compressing files, compressed and uncompressed sizes + SHOULD be stored in ZIP64 format (as 8 byte values) when a + file's size exceeds 0xFFFFFFFF. However ZIP64 format MAY be + used regardless of the size of a file. When extracting, if + the zip64 extended information extra field is present for + the file the compressed and uncompressed sizes will be 8 + byte values. + + 4.3.9.3 Although not originally assigned a signature, the value + 0x08074b50 has commonly been adopted as a signature value + for the data descriptor record. Implementers SHOULD be + aware that ZIP files MAY be encountered with or without this + signature marking data descriptors and SHOULD account for + either case when reading ZIP files to ensure compatibility. + + 4.3.9.4 When writing ZIP files, implementors SHOULD include the + signature value marking the data descriptor record. When + the signature is used, the fields currently defined for + the data descriptor record will immediately follow the + signature. + + 4.3.9.5 An extensible data descriptor will be released in a + future version of this APPNOTE. This new record is intended to + resolve conflicts with the use of this record going forward, + and to provide better support for streamed file processing. + + 4.3.9.6 When the Central Directory Encryption method is used, + the data descriptor record is not required, but MAY be used. + If present, and bit 3 of the general purpose bit field is set to + indicate its presence, the values in fields of the data descriptor + record MUST be set to binary zeros. See the section on the Strong + Encryption Specification for information. Refer to the section in + this document entitled "Incorporating PKWARE Proprietary Technology + into Your Product" for more information. + + + 4.3.10 Archive decryption header: + + 4.3.10.1 The Archive Decryption Header is introduced in version 6.2 + of the ZIP format specification. This record exists in support + of the Central Directory Encryption Feature implemented as part of + the Strong Encryption Specification as described in this document. + When the Central Directory Structure is encrypted, this decryption + header MUST precede the encrypted data segment. + + 4.3.10.2 The encrypted data segment SHALL consist of the Archive + extra data record (if present) and the encrypted Central Directory + Structure data. The format of this data record is identical to the + Decryption header record preceding compressed file data. If the + central directory structure is encrypted, the location of the start of + this data record is determined using the Start of Central Directory + field in the Zip64 End of Central Directory record. See the + section on the Strong Encryption Specification for information + on the fields used in the Archive Decryption Header record. + Refer to the section in this document entitled "Incorporating + PKWARE Proprietary Technology into Your Product" for more information. + + + 4.3.11 Archive extra data record: + + archive extra data signature 4 bytes (0x08064b50) + extra field length 4 bytes + extra field data (variable size) + + 4.3.11.1 The Archive Extra Data Record is introduced in version 6.2 + of the ZIP format specification. This record MAY be used in support + of the Central Directory Encryption Feature implemented as part of + the Strong Encryption Specification as described in this document. + When present, this record MUST immediately precede the central + directory data structure. + + 4.3.11.2 The size of this data record SHALL be included in the + Size of the Central Directory field in the End of Central + Directory record. If the central directory structure is compressed, + but not encrypted, the location of the start of this data record is + determined using the Start of Central Directory field in the Zip64 + End of Central Directory record. Refer to the section in this document + entitled "Incorporating PKWARE Proprietary Technology into Your + Product" for more information. + + 4.3.12 Central directory structure: + + [central directory header 1] + . + . + . + [central directory header n] + [digital signature] + + File header: + + central file header signature 4 bytes (0x02014b50) + version made by 2 bytes + version needed to extract 2 bytes + general purpose bit flag 2 bytes + compression method 2 bytes + last mod file time 2 bytes + last mod file date 2 bytes + crc-32 4 bytes + compressed size 4 bytes + uncompressed size 4 bytes + file name length 2 bytes + extra field length 2 bytes + file comment length 2 bytes + disk number start 2 bytes + internal file attributes 2 bytes + external file attributes 4 bytes + relative offset of local header 4 bytes + + file name (variable size) + extra field (variable size) + file comment (variable size) + + 4.3.13 Digital signature: + + header signature 4 bytes (0x05054b50) + size of data 2 bytes + signature data (variable size) + + With the introduction of the Central Directory Encryption + feature in version 6.2 of this specification, the Central + Directory Structure MAY be stored both compressed and encrypted. + Although not required, it is assumed when encrypting the + Central Directory Structure, that it will be compressed + for greater storage efficiency. Information on the + Central Directory Encryption feature can be found in the section + describing the Strong Encryption Specification. The Digital + Signature record will be neither compressed nor encrypted. + + 4.3.14 Zip64 end of central directory record + + zip64 end of central dir + signature 4 bytes (0x06064b50) + size of zip64 end of central + directory record 8 bytes + version made by 2 bytes + version needed to extract 2 bytes + number of this disk 4 bytes + number of the disk with the + start of the central directory 4 bytes + total number of entries in the + central directory on this disk 8 bytes + total number of entries in the + central directory 8 bytes + size of the central directory 8 bytes + offset of start of central + directory with respect to + the starting disk number 8 bytes + zip64 extensible data sector (variable size) + + 4.3.14.1 The value stored into the "size of zip64 end of central + directory record" SHOULD be the size of the remaining + record and SHOULD NOT include the leading 12 bytes. + + Size = SizeOfFixedFields + SizeOfVariableData - 12. + + 4.3.14.2 The above record structure defines Version 1 of the + zip64 end of central directory record. Version 1 was + implemented in versions of this specification preceding + 6.2 in support of the ZIP64 large file feature. The + introduction of the Central Directory Encryption feature + implemented in version 6.2 as part of the Strong Encryption + Specification defines Version 2 of this record structure. + Refer to the section describing the Strong Encryption + Specification for details on the version 2 format for + this record. Refer to the section in this document entitled + "Incorporating PKWARE Proprietary Technology into Your Product" + for more information applicable to use of Version 2 of this + record. + + 4.3.14.3 Special purpose data MAY reside in the zip64 extensible + data sector field following either a V1 or V2 version of this + record. To ensure identification of this special purpose data + it MUST include an identifying header block consisting of the + following: + + Header ID - 2 bytes + Data Size - 4 bytes + + The Header ID field indicates the type of data that is in the + data block that follows. + + Data Size identifies the number of bytes that follow for this + data block type. + + 4.3.14.4 Multiple special purpose data blocks MAY be present. + Each MUST be preceded by a Header ID and Data Size field. Current + mappings of Header ID values supported in this field are as + defined in APPENDIX C. + + 4.3.15 Zip64 end of central directory locator + + zip64 end of central dir locator + signature 4 bytes (0x07064b50) + number of the disk with the + start of the zip64 end of + central directory 4 bytes + relative offset of the zip64 + end of central directory record 8 bytes + total number of disks 4 bytes + + 4.3.16 End of central directory record: + + end of central dir signature 4 bytes (0x06054b50) + number of this disk 2 bytes + number of the disk with the + start of the central directory 2 bytes + total number of entries in the + central directory on this disk 2 bytes + total number of entries in + the central directory 2 bytes + size of the central directory 4 bytes + offset of start of central + directory with respect to + the starting disk number 4 bytes + .ZIP file comment length 2 bytes + .ZIP file comment (variable size) + +4.4 Explanation of fields +-------------------------- + + 4.4.1 General notes on fields + + 4.4.1.1 All fields unless otherwise noted are unsigned and stored + in Intel low-byte:high-byte, low-word:high-word order. + + 4.4.1.2 String fields are not null terminated, since the length + is given explicitly. + + 4.4.1.3 The entries in the central directory MAY NOT necessarily + be in the same order that files appear in the .ZIP file. + + 4.4.1.4 If one of the fields in the end of central directory + record is too small to hold required data, the field SHOULD be + set to -1 (0xFFFF or 0xFFFFFFFF) and the ZIP64 format record + SHOULD be created. + + 4.4.1.5 The end of central directory record and the Zip64 end + of central directory locator record MUST reside on the same + disk when splitting or spanning an archive. + + 4.4.2 version made by (2 bytes) + + 4.4.2.1 The upper byte indicates the compatibility of the file + attribute information. If the external file attributes + are compatible with MS-DOS and can be read by PKZIP for + DOS version 2.04g then this value will be zero. If these + attributes are not compatible, then this value will + identify the host system on which the attributes are + compatible. Software can use this information to determine + the line record format for text files etc. + + 4.4.2.2 The current mappings are: + + 0 - MS-DOS and OS/2 (FAT / VFAT / FAT32 file systems) + 1 - Amiga 2 - OpenVMS + 3 - UNIX 4 - VM/CMS + 5 - Atari ST 6 - OS/2 H.P.F.S. + 7 - Macintosh 8 - Z-System + 9 - CP/M 10 - Windows NTFS + 11 - MVS (OS/390 - Z/OS) 12 - VSE + 13 - Acorn Risc 14 - VFAT + 15 - alternate MVS 16 - BeOS + 17 - Tandem 18 - OS/400 + 19 - OS X (Darwin) 20 thru 255 - unused + + 4.4.2.3 The lower byte indicates the ZIP specification version + (the version of this document) supported by the software + used to encode the file. The value/10 indicates the major + version number, and the value mod 10 is the minor version + number. + + 4.4.3 version needed to extract (2 bytes) + + 4.4.3.1 The minimum supported ZIP specification version needed + to extract the file, mapped as above. This value is based on + the specific format features a ZIP program MUST support to + be able to extract the file. If multiple features are + applied to a file, the minimum version MUST be set to the + feature having the highest value. New features or feature + changes affecting the published format specification will be + implemented using higher version numbers than the last + published value to avoid conflict. + + 4.4.3.2 Current minimum feature versions are as defined below: + + 1.0 - Default value + 1.1 - File is a volume label + 2.0 - File is a folder (directory) + 2.0 - File is compressed using Deflate compression + 2.0 - File is encrypted using traditional PKWARE encryption + 2.1 - File is compressed using Deflate64(tm) + 2.5 - File is compressed using PKWARE DCL Implode + 2.7 - File is a patch data set + 4.5 - File uses ZIP64 format extensions + 4.6 - File is compressed using BZIP2 compression* + 5.0 - File is encrypted using DES + 5.0 - File is encrypted using 3DES + 5.0 - File is encrypted using original RC2 encryption + 5.0 - File is encrypted using RC4 encryption + 5.1 - File is encrypted using AES encryption + 5.1 - File is encrypted using corrected RC2 encryption** + 5.2 - File is encrypted using corrected RC2-64 encryption** + 6.1 - File is encrypted using non-OAEP key wrapping*** + 6.2 - Central directory encryption + 6.3 - File is compressed using LZMA + 6.3 - File is compressed using PPMd+ + 6.3 - File is encrypted using Blowfish + 6.3 - File is encrypted using Twofish + + 4.4.3.3 Notes on version needed to extract + + * Early 7.x (pre-7.2) versions of PKZIP incorrectly set the + version needed to extract for BZIP2 compression to be 50 + when it SHOULD have been 46. + + ** Refer to the section on Strong Encryption Specification + for additional information regarding RC2 corrections. + + *** Certificate encryption using non-OAEP key wrapping is the + intended mode of operation for all versions beginning with 6.1. + Support for OAEP key wrapping MUST only be used for + backward compatibility when sending ZIP files to be opened by + versions of PKZIP older than 6.1 (5.0 or 6.0). + + + Files compressed using PPMd MUST set the version + needed to extract field to 6.3, however, not all ZIP + programs enforce this and MAY be unable to decompress + data files compressed using PPMd if this value is set. + + When using ZIP64 extensions, the corresponding value in the + zip64 end of central directory record MUST also be set. + This field SHOULD be set appropriately to indicate whether + Version 1 or Version 2 format is in use. + + + 4.4.4 general purpose bit flag: (2 bytes) + + Bit 0: If set, indicates that the file is encrypted. + + (For Method 6 - Imploding) + Bit 1: If the compression method used was type 6, + Imploding, then this bit, if set, indicates + an 8K sliding dictionary was used. If clear, + then a 4K sliding dictionary was used. + + Bit 2: If the compression method used was type 6, + Imploding, then this bit, if set, indicates + 3 Shannon-Fano trees were used to encode the + sliding dictionary output. If clear, then 2 + Shannon-Fano trees were used. + + (For Methods 8 and 9 - Deflating) + Bit 2 Bit 1 + 0 0 Normal (-en) compression option was used. + 0 1 Maximum (-exx/-ex) compression option was used. + 1 0 Fast (-ef) compression option was used. + 1 1 Super Fast (-es) compression option was used. + + (For Method 14 - LZMA) + Bit 1: If the compression method used was type 14, + LZMA, then this bit, if set, indicates + an end-of-stream (EOS) marker is used to + mark the end of the compressed data stream. + If clear, then an EOS marker is not present + and the compressed data size must be known + to extract. + + Note: Bits 1 and 2 are undefined if the compression + method is any other. + + Bit 3: If this bit is set, the fields crc-32, compressed + size and uncompressed size are set to zero in the + local header. The correct values are put in the + data descriptor immediately following the compressed + data. (Note: PKZIP version 2.04g for DOS only + recognizes this bit for method 8 compression, newer + versions of PKZIP recognize this bit for any + compression method.) + + Bit 4: Reserved for use with method 8, for enhanced + deflating. + + Bit 5: If this bit is set, this indicates that the file is + compressed patched data. (Note: Requires PKZIP + version 2.70 or greater) + + Bit 6: Strong encryption. If this bit is set, you MUST + set the version needed to extract value to at least + 50 and you MUST also set bit 0. If AES encryption + is used, the version needed to extract value MUST + be at least 51. See the section describing the Strong + Encryption Specification for details. Refer to the + section in this document entitled "Incorporating PKWARE + Proprietary Technology into Your Product" for more + information. + + Bit 7: Currently unused. + + Bit 8: Currently unused. + + Bit 9: Currently unused. + + Bit 10: Currently unused. + + Bit 11: Language encoding flag (EFS). If this bit is set, + the filename and comment fields for this file + MUST be encoded using UTF-8. (see APPENDIX D) + + Bit 12: Reserved by PKWARE for enhanced compression. + + Bit 13: Set when encrypting the Central Directory to indicate + selected data values in the Local Header are masked to + hide their actual values. See the section describing + the Strong Encryption Specification for details. Refer + to the section in this document entitled "Incorporating + PKWARE Proprietary Technology into Your Product" for + more information. + + Bit 14: Reserved by PKWARE for alternate streams. + + Bit 15: Reserved by PKWARE. + + 4.4.5 compression method: (2 bytes) + + 0 - The file is stored (no compression) + 1 - The file is Shrunk + 2 - The file is Reduced with compression factor 1 + 3 - The file is Reduced with compression factor 2 + 4 - The file is Reduced with compression factor 3 + 5 - The file is Reduced with compression factor 4 + 6 - The file is Imploded + 7 - Reserved for Tokenizing compression algorithm + 8 - The file is Deflated + 9 - Enhanced Deflating using Deflate64(tm) + 10 - PKWARE Data Compression Library Imploding (old IBM TERSE) + 11 - Reserved by PKWARE + 12 - File is compressed using BZIP2 algorithm + 13 - Reserved by PKWARE + 14 - LZMA + 15 - Reserved by PKWARE + 16 - IBM z/OS CMPSC Compression + 17 - Reserved by PKWARE + 18 - File is compressed using IBM TERSE (new) + 19 - IBM LZ77 z Architecture + 20 - deprecated (use method 93 for zstd) + 93 - Zstandard (zstd) Compression + 94 - MP3 Compression + 95 - XZ Compression + 96 - JPEG variant + 97 - WavPack compressed data + 98 - PPMd version I, Rev 1 + 99 - AE-x encryption marker (see APPENDIX E) + + 4.4.5.1 Methods 1-6 are legacy algorithms and are no longer + recommended for use when compressing files. + + 4.4.6 date and time fields: (2 bytes each) + + The date and time are encoded in standard MS-DOS format. + If input came from standard input, the date and time are + those at which compression was started for this data. + If encrypting the central directory and general purpose bit + flag 13 is set indicating masking, the value stored in the + Local Header will be zero. MS-DOS time format is different + from more commonly used computer time formats such as + UTC. For example, MS-DOS uses year values relative to 1980 + and 2 second precision. + + 4.4.7 CRC-32: (4 bytes) + + The CRC-32 algorithm was generously contributed by + David Schwaderer and can be found in his excellent + book "C Programmers Guide to NetBIOS" published by + Howard W. Sams & Co. Inc. The 'magic number' for + the CRC is 0xdebb20e3. The proper CRC pre and post + conditioning is used, meaning that the CRC register + is pre-conditioned with all ones (a starting value + of 0xffffffff) and the value is post-conditioned by + taking the one's complement of the CRC residual. + If bit 3 of the general purpose flag is set, this + field is set to zero in the local header and the correct + value is put in the data descriptor and in the central + directory. When encrypting the central directory, if the + local header is not in ZIP64 format and general purpose + bit flag 13 is set indicating masking, the value stored + in the Local Header will be zero. + + 4.4.8 compressed size: (4 bytes) + 4.4.9 uncompressed size: (4 bytes) + + The size of the file compressed (4.4.8) and uncompressed, + (4.4.9) respectively. When a decryption header is present it + will be placed in front of the file data and the value of the + compressed file size will include the bytes of the decryption + header. If bit 3 of the general purpose bit flag is set, + these fields are set to zero in the local header and the + correct values are put in the data descriptor and + in the central directory. If an archive is in ZIP64 format + and the value in this field is 0xFFFFFFFF, the size will be + in the corresponding 8 byte ZIP64 extended information + extra field. When encrypting the central directory, if the + local header is not in ZIP64 format and general purpose bit + flag 13 is set indicating masking, the value stored for the + uncompressed size in the Local Header will be zero. + + 4.4.10 file name length: (2 bytes) + 4.4.11 extra field length: (2 bytes) + 4.4.12 file comment length: (2 bytes) + + The length of the file name, extra field, and comment + fields respectively. The combined length of any + directory record and these three fields SHOULD NOT + generally exceed 65,535 bytes. If input came from standard + input, the file name length is set to zero. + + + 4.4.13 disk number start: (2 bytes) + + The number of the disk on which this file begins. If an + archive is in ZIP64 format and the value in this field is + 0xFFFF, the size will be in the corresponding 4 byte zip64 + extended information extra field. + + 4.4.14 internal file attributes: (2 bytes) + + Bits 1 and 2 are reserved for use by PKWARE. + + 4.4.14.1 The lowest bit of this field indicates, if set, + that the file is apparently an ASCII or text file. If not + set, that the file apparently contains binary data. + The remaining bits are unused in version 1.0. + + 4.4.14.2 The 0x0002 bit of this field indicates, if set, that + a 4 byte variable record length control field precedes each + logical record indicating the length of the record. The + record length control field is stored in little-endian byte + order. This flag is independent of text control characters, + and if used in conjunction with text data, includes any + control characters in the total length of the record. This + value is provided for mainframe data transfer support. + + 4.4.15 external file attributes: (4 bytes) + + The mapping of the external attributes is + host-system dependent (see 'version made by'). For + MS-DOS, the low order byte is the MS-DOS directory + attribute byte. If input came from standard input, this + field is set to zero. + + 4.4.16 relative offset of local header: (4 bytes) + + This is the offset from the start of the first disk on + which this file appears, to where the local header SHOULD + be found. If an archive is in ZIP64 format and the value + in this field is 0xFFFFFFFF, the size will be in the + corresponding 8 byte zip64 extended information extra field. + + 4.4.17 file name: (Variable) + + 4.4.17.1 The name of the file, with optional relative path. + The path stored MUST NOT contain a drive or + device letter, or a leading slash. All slashes + MUST be forward slashes '/' as opposed to + backwards slashes '\' for compatibility with Amiga + and UNIX file systems etc. If input came from standard + input, there is no file name field. + + 4.4.17.2 If using the Central Directory Encryption Feature and + general purpose bit flag 13 is set indicating masking, the file + name stored in the Local Header will not be the actual file name. + A masking value consisting of a unique hexadecimal value will + be stored. This value will be sequentially incremented for each + file in the archive. See the section on the Strong Encryption + Specification for details on retrieving the encrypted file name. + Refer to the section in this document entitled "Incorporating PKWARE + Proprietary Technology into Your Product" for more information. + + + 4.4.18 file comment: (Variable) + + The comment for this file. + + 4.4.19 number of this disk: (2 bytes) + + The number of this disk, which contains central + directory end record. If an archive is in ZIP64 format + and the value in this field is 0xFFFF, the size will + be in the corresponding 4 byte zip64 end of central + directory field. + + + 4.4.20 number of the disk with the start of the central + directory: (2 bytes) + + The number of the disk on which the central + directory starts. If an archive is in ZIP64 format + and the value in this field is 0xFFFF, the size will + be in the corresponding 4 byte zip64 end of central + directory field. + + 4.4.21 total number of entries in the central dir on + this disk: (2 bytes) + + The number of central directory entries on this disk. + If an archive is in ZIP64 format and the value in + this field is 0xFFFF, the size will be in the + corresponding 8 byte zip64 end of central + directory field. + + 4.4.22 total number of entries in the central dir: (2 bytes) + + The total number of files in the .ZIP file. If an + archive is in ZIP64 format and the value in this field + is 0xFFFF, the size will be in the corresponding 8 byte + zip64 end of central directory field. + + 4.4.23 size of the central directory: (4 bytes) + + The size (in bytes) of the entire central directory. + If an archive is in ZIP64 format and the value in + this field is 0xFFFFFFFF, the size will be in the + corresponding 8 byte zip64 end of central + directory field. + + 4.4.24 offset of start of central directory with respect to + the starting disk number: (4 bytes) + + Offset of the start of the central directory on the + disk on which the central directory starts. If an + archive is in ZIP64 format and the value in this + field is 0xFFFFFFFF, the size will be in the + corresponding 8 byte zip64 end of central + directory field. + + 4.4.25 .ZIP file comment length: (2 bytes) + + The length of the comment for this .ZIP file. + + 4.4.26 .ZIP file comment: (Variable) + + The comment for this .ZIP file. ZIP file comment data + is stored unsecured. No encryption or data authentication + is applied to this area at this time. Confidential information + SHOULD NOT be stored in this section. + + 4.4.27 zip64 extensible data sector (variable size) + + (currently reserved for use by PKWARE) + + + 4.4.28 extra field: (Variable) + + This SHOULD be used for storage expansion. If additional + information needs to be stored within a ZIP file for special + application or platform needs, it SHOULD be stored here. + Programs supporting earlier versions of this specification can + then safely skip the file, and find the next file or header. + This field will be 0 length in version 1.0. + + Existing extra fields are defined in the section + Extensible data fields that follows. + +4.5 Extensible data fields +-------------------------- + + 4.5.1 In order to allow different programs and different types + of information to be stored in the 'extra' field in .ZIP + files, the following structure MUST be used for all + programs storing data in this field: + + header1+data1 + header2+data2 . . . + + Each header MUST consist of: + + Header ID - 2 bytes + Data Size - 2 bytes + + Note: all fields stored in Intel low-byte/high-byte order. + + The Header ID field indicates the type of data that is in + the following data block. + + Header IDs of 0 thru 31 are reserved for use by PKWARE. + The remaining IDs can be used by third party vendors for + proprietary usage. + + 4.5.2 The current Header ID mappings defined by PKWARE are: + + 0x0001 Zip64 extended information extra field + 0x0007 AV Info + 0x0008 Reserved for extended language encoding data (PFS) + (see APPENDIX D) + 0x0009 OS/2 + 0x000a NTFS + 0x000c OpenVMS + 0x000d UNIX + 0x000e Reserved for file stream and fork descriptors + 0x000f Patch Descriptor + 0x0014 PKCS#7 Store for X.509 Certificates + 0x0015 X.509 Certificate ID and Signature for + individual file + 0x0016 X.509 Certificate ID for Central Directory + 0x0017 Strong Encryption Header + 0x0018 Record Management Controls + 0x0019 PKCS#7 Encryption Recipient Certificate List + 0x0020 Reserved for Timestamp record + 0x0021 Policy Decryption Key Record + 0x0022 Smartcrypt Key Provider Record + 0x0023 Smartcrypt Policy Key Data Record + 0x0065 IBM S/390 (Z390), AS/400 (I400) attributes + - uncompressed + 0x0066 Reserved for IBM S/390 (Z390), AS/400 (I400) + attributes - compressed + 0x4690 POSZIP 4690 (reserved) + + + 4.5.3 -Zip64 Extended Information Extra Field (0x0001): + + The following is the layout of the zip64 extended + information "extra" block. If one of the size or + offset fields in the Local or Central directory + record is too small to hold the required data, + a Zip64 extended information record is created. + The order of the fields in the zip64 extended + information record is fixed, but the fields MUST + only appear if the corresponding Local or Central + directory record field is set to 0xFFFF or 0xFFFFFFFF. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(ZIP64) 0x0001 2 bytes Tag for this "extra" block type + Size 2 bytes Size of this "extra" block + Original + Size 8 bytes Original uncompressed file size + Compressed + Size 8 bytes Size of compressed data + Relative Header + Offset 8 bytes Offset of local header record + Disk Start + Number 4 bytes Number of the disk on which + this file starts + + This entry in the Local header MUST include BOTH original + and compressed file size fields. If encrypting the + central directory and bit 13 of the general purpose bit + flag is set indicating masking, the value stored in the + Local Header for the original file size will be zero. + + + 4.5.4 -OS/2 Extra Field (0x0009): + + The following is the layout of the OS/2 attributes "extra" + block. (Last Revision 09/05/95) + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(OS/2) 0x0009 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + BSize 4 bytes Uncompressed Block Size + CType 2 bytes Compression type + EACRC 4 bytes CRC value for uncompress block + (var) variable Compressed block + + The OS/2 extended attribute structure (FEA2LIST) is + compressed and then stored in its entirety within this + structure. There will only ever be one "block" of data in + VarFields[]. + + 4.5.5 -NTFS Extra Field (0x000a): + + The following is the layout of the NTFS attributes + "extra" block. (Note: At this time the Mtime, Atime + and Ctime values MAY be used on any WIN32 system.) + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(NTFS) 0x000a 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of the total "extra" block + Reserved 4 bytes Reserved for future use + Tag1 2 bytes NTFS attribute tag value #1 + Size1 2 bytes Size of attribute #1, in bytes + (var) Size1 Attribute #1 data + . + . + . + TagN 2 bytes NTFS attribute tag value #N + SizeN 2 bytes Size of attribute #N, in bytes + (var) SizeN Attribute #N data + + For NTFS, values for Tag1 through TagN are as follows: + (currently only one set of attributes is defined for NTFS) + + Tag Size Description + ----- ---- ----------- + 0x0001 2 bytes Tag for attribute #1 + Size1 2 bytes Size of attribute #1, in bytes + Mtime 8 bytes File last modification time + Atime 8 bytes File last access time + Ctime 8 bytes File creation time + + 4.5.6 -OpenVMS Extra Field (0x000c): + + The following is the layout of the OpenVMS attributes + "extra" block. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- + (VMS) 0x000c 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of the total "extra" block + CRC 4 bytes 32-bit CRC for remainder of the block + Tag1 2 bytes OpenVMS attribute tag value #1 + Size1 2 bytes Size of attribute #1, in bytes + (var) Size1 Attribute #1 data + . + . + . + TagN 2 bytes OpenVMS attribute tag value #N + SizeN 2 bytes Size of attribute #N, in bytes + (var) SizeN Attribute #N data + + OpenVMS Extra Field Rules: + + 4.5.6.1. There will be one or more attributes present, which + will each be preceded by the above TagX & SizeX values. + These values are identical to the ATR$C_XXXX and ATR$S_XXXX + constants which are defined in ATR.H under OpenVMS C. Neither + of these values will ever be zero. + + 4.5.6.2. No word alignment or padding is performed. + + 4.5.6.3. A well-behaved PKZIP/OpenVMS program SHOULD NOT produce + more than one sub-block with the same TagX value. Also, there MUST + NOT be more than one "extra" block of type 0x000c in a particular + directory record. + + 4.5.7 -UNIX Extra Field (0x000d): + + The following is the layout of the UNIX "extra" block. + Note: all fields are stored in Intel low-byte/high-byte + order. + + Value Size Description + ----- ---- ----------- +(UNIX) 0x000d 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + Atime 4 bytes File last access time + Mtime 4 bytes File last modification time + Uid 2 bytes File user ID + Gid 2 bytes File group ID + (var) variable Variable length data field + + The variable length data field will contain file type + specific data. Currently the only values allowed are + the original "linked to" file names for hard or symbolic + links, and the major and minor device node numbers for + character and block device nodes. Since device nodes + cannot be either symbolic or hard links, only one set of + variable length data is stored. Link files will have the + name of the original file stored. This name is NOT NULL + terminated. Its size can be determined by checking TSize - + 12. Device entries will have eight bytes stored as two 4 + byte entries (in little endian format). The first entry + will be the major device number, and the second the minor + device number. + + 4.5.8 -PATCH Descriptor Extra Field (0x000f): + + 4.5.8.1 The following is the layout of the Patch Descriptor + "extra" block. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(Patch) 0x000f 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of the total "extra" block + Version 2 bytes Version of the descriptor + Flags 4 bytes Actions and reactions (see below) + OldSize 4 bytes Size of the file about to be patched + OldCRC 4 bytes 32-bit CRC of the file to be patched + NewSize 4 bytes Size of the resulting file + NewCRC 4 bytes 32-bit CRC of the resulting file + + 4.5.8.2 Actions and reactions + + Bits Description + ---- ---------------- + 0 Use for auto detection + 1 Treat as a self-patch + 2-3 RESERVED + 4-5 Action (see below) + 6-7 RESERVED + 8-9 Reaction (see below) to absent file + 10-11 Reaction (see below) to newer file + 12-13 Reaction (see below) to unknown file + 14-15 RESERVED + 16-31 RESERVED + + 4.5.8.2.1 Actions + + Action Value + ------ ----- + none 0 + add 1 + delete 2 + patch 3 + + 4.5.8.2.2 Reactions + + Reaction Value + -------- ----- + ask 0 + skip 1 + ignore 2 + fail 3 + + 4.5.8.3 Patch support is provided by PKPatchMaker(tm) technology + and is covered under U.S. Patents and Patents Pending. The use or + implementation in a product of certain technological aspects set + forth in the current APPNOTE, including those with regard to + strong encryption or patching requires a license from PKWARE. + Refer to the section in this document entitled "Incorporating + PKWARE Proprietary Technology into Your Product" for more + information. + + 4.5.9 -PKCS#7 Store for X.509 Certificates (0x0014): + + This field MUST contain information about each of the certificates + files MAY be signed with. When the Central Directory Encryption + feature is enabled for a ZIP file, this record will appear in + the Archive Extra Data Record, otherwise it will appear in the + first central directory record and will be ignored in any + other record. + + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(Store) 0x0014 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of the store data + TData TSize Data about the store + + + 4.5.10 -X.509 Certificate ID and Signature for individual file (0x0015): + + This field contains the information about which certificate in + the PKCS#7 store was used to sign a particular file. It also + contains the signature data. This field can appear multiple + times, but can only appear once per certificate. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(CID) 0x0015 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of data that follows + TData TSize Signature Data + + 4.5.11 -X.509 Certificate ID and Signature for central directory (0x0016): + + This field contains the information about which certificate in + the PKCS#7 store was used to sign the central directory structure. + When the Central Directory Encryption feature is enabled for a + ZIP file, this record will appear in the Archive Extra Data Record, + otherwise it will appear in the first central directory record. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(CDID) 0x0016 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of data that follows + TData TSize Data + + 4.5.12 -Strong Encryption Header (0x0017): + + Value Size Description + ----- ---- ----------- + 0x0017 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of data that follows + Format 2 bytes Format definition for this record + AlgID 2 bytes Encryption algorithm identifier + Bitlen 2 bytes Bit length of encryption key + Flags 2 bytes Processing flags + CertData TSize-8 Certificate decryption extra field data + (refer to the explanation for CertData + in the section describing the + Certificate Processing Method under + the Strong Encryption Specification) + + See the section describing the Strong Encryption Specification + for details. Refer to the section in this document entitled + "Incorporating PKWARE Proprietary Technology into Your Product" + for more information. + + 4.5.13 -Record Management Controls (0x0018): + + Value Size Description + ----- ---- ----------- +(Rec-CTL) 0x0018 2 bytes Tag for this "extra" block type + CSize 2 bytes Size of total extra block data + Tag1 2 bytes Record control attribute 1 + Size1 2 bytes Size of attribute 1, in bytes + Data1 Size1 Attribute 1 data + . + . + . + TagN 2 bytes Record control attribute N + SizeN 2 bytes Size of attribute N, in bytes + DataN SizeN Attribute N data + + + 4.5.14 -PKCS#7 Encryption Recipient Certificate List (0x0019): + + This field MAY contain information about each of the certificates + used in encryption processing and it can be used to identify who is + allowed to decrypt encrypted files. This field SHOULD only appear + in the archive extra data record. This field is not required and + serves only to aid archive modifications by preserving public + encryption key data. Individual security requirements may dictate + that this data be omitted to deter information exposure. + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- +(CStore) 0x0019 2 bytes Tag for this "extra" block type + TSize 2 bytes Size of the store data + TData TSize Data about the store + + TData: + + Value Size Description + ----- ---- ----------- + Version 2 bytes Format version number - MUST be 0x0001 at this time + CStore (var) PKCS#7 data blob + + See the section describing the Strong Encryption Specification + for details. Refer to the section in this document entitled + "Incorporating PKWARE Proprietary Technology into Your Product" + for more information. + + 4.5.15 -MVS Extra Field (0x0065): + + The following is the layout of the MVS "extra" block. + Note: Some fields are stored in Big Endian format. + All text is in EBCDIC format unless otherwise specified. +Value Size Description + ----- ---- ----------- +(MVS) 0x0065 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + ID 4 bytes EBCDIC "Z390" 0xE9F3F9F0 or + "T4MV" for TargetFour + (var) TSize-4 Attribute data (see APPENDIX B) + + + 4.5.16 -OS/400 Extra Field (0x0065): + + The following is the layout of the OS/400 "extra" block. + Note: Some fields are stored in Big Endian format. + All text is in EBCDIC format unless otherwise specified. + + Value Size Description + ----- ---- ----------- +(OS400) 0x0065 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + ID 4 bytes EBCDIC "I400" 0xC9F4F0F0 or + "T4MV" for TargetFour + (var) TSize-4 Attribute data (see APPENDIX A) + + 4.5.17 -Policy Decryption Key Record Extra Field (0x0021): + + The following is the layout of the Policy Decryption Key "extra" block. + TData is a variable length, variable content field. It holds + information about encryptions and/or encryption key sources. + Contact PKWARE for information on current TData structures. + Information in this "extra" block may aternatively be placed + within comment fields. Refer to the section in this document + entitled "Incorporating PKWARE Proprietary Technology into Your + Product" for more information. + + Value Size Description + ----- ---- ----------- + 0x0021 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + TData TSize Data about the key + + 4.5.18 -Key Provider Record Extra Field (0x0022): + + The following is the layout of the Key Provider "extra" block. + TData is a variable length, variable content field. It holds + information about encryptions and/or encryption key sources. + Contact PKWARE for information on current TData structures. + Information in this "extra" block may aternatively be placed + within comment fields. Refer to the section in this document + entitled "Incorporating PKWARE Proprietary Technology into Your + Product" for more information. + + Value Size Description + ----- ---- ----------- + 0x0022 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + TData TSize Data about the key + + 4.5.19 -Policy Key Data Record Record Extra Field (0x0023): + + The following is the layout of the Policy Key Data "extra" block. + TData is a variable length, variable content field. It holds + information about encryptions and/or encryption key sources. + Contact PKWARE for information on current TData structures. + Information in this "extra" block may aternatively be placed + within comment fields. Refer to the section in this document + entitled "Incorporating PKWARE Proprietary Technology into Your + Product" for more information. + + Value Size Description + ----- ---- ----------- + 0x0023 2 bytes Tag for this "extra" block type + TSize 2 bytes Size for the following data block + TData TSize Data about the key + +4.6 Third Party Mappings +------------------------ + + 4.6.1 Third party mappings commonly used are: + + 0x07c8 Macintosh + 0x2605 ZipIt Macintosh + 0x2705 ZipIt Macintosh 1.3.5+ + 0x2805 ZipIt Macintosh 1.3.5+ + 0x334d Info-ZIP Macintosh + 0x4341 Acorn/SparkFS + 0x4453 Windows NT security descriptor (binary ACL) + 0x4704 VM/CMS + 0x470f MVS + 0x4b46 FWKCS MD5 (see below) + 0x4c41 OS/2 access control list (text ACL) + 0x4d49 Info-ZIP OpenVMS + 0x4f4c Xceed original location extra field + 0x5356 AOS/VS (ACL) + 0x5455 extended timestamp + 0x554e Xceed unicode extra field + 0x5855 Info-ZIP UNIX (original, also OS/2, NT, etc) + 0x6375 Info-ZIP Unicode Comment Extra Field + 0x6542 BeOS/BeBox + 0x7075 Info-ZIP Unicode Path Extra Field + 0x756e ASi UNIX + 0x7855 Info-ZIP UNIX (new) + 0xa11e Data Stream Alignment (Apache Commons-Compress) + 0xa220 Microsoft Open Packaging Growth Hint + 0xfd4a SMS/QDOS + 0x9901 AE-x encryption structure (see APPENDIX E) + 0x9902 unknown + + + Detailed descriptions of Extra Fields defined by third + party mappings will be documented as information on + these data structures is made available to PKWARE. + PKWARE does not guarantee the accuracy of any published + third party data. + + 4.6.2 Third-party Extra Fields MUST include a Header ID using + the format defined in the section of this document + titled Extensible Data Fields (section 4.5). + + The Data Size field indicates the size of the following + data block. Programs can use this value to skip to the + next header block, passing over any data blocks that are + not of interest. + + Note: As stated above, the size of the entire .ZIP file + header, including the file name, comment, and extra + field SHOULD NOT exceed 64K in size. + + 4.6.3 In case two different programs appropriate the same + Header ID value, it is strongly recommended that each + program SHOULD place a unique signature of at least two bytes in + size (and preferably 4 bytes or bigger) at the start of + each data area. Every program SHOULD verify that its + unique signature is present, in addition to the Header ID + value being correct, before assuming that it is a block of + known type. + + Third-party Mappings: + + 4.6.4 -ZipIt Macintosh Extra Field (long) (0x2605): + + The following is the layout of the ZipIt extra block + for Macintosh. The local-header and central-header versions + are identical. This block MUST be present if the file is + stored MacBinary-encoded and it SHOULD NOT be used if the file + is not stored MacBinary-encoded. + + Value Size Description + ----- ---- ----------- + (Mac2) 0x2605 Short tag for this extra block type + TSize Short total data size for this block + "ZPIT" beLong extra-field signature + FnLen Byte length of FileName + FileName variable full Macintosh filename + FileType Byte[4] four-byte Mac file type string + Creator Byte[4] four-byte Mac creator string + + + 4.6.5 -ZipIt Macintosh Extra Field (short, for files) (0x2705): + + The following is the layout of a shortened variant of the + ZipIt extra block for Macintosh (without "full name" entry). + This variant is used by ZipIt 1.3.5 and newer for entries of + files (not directories) that do not have a MacBinary encoded + file. The local-header and central-header versions are identical. + + Value Size Description + ----- ---- ----------- + (Mac2b) 0x2705 Short tag for this extra block type + TSize Short total data size for this block (12) + "ZPIT" beLong extra-field signature + FileType Byte[4] four-byte Mac file type string + Creator Byte[4] four-byte Mac creator string + fdFlags beShort attributes from FInfo.frFlags, + MAY be omitted + 0x0000 beShort reserved, MAY be omitted + + + 4.6.6 -ZipIt Macintosh Extra Field (short, for directories) (0x2805): + + The following is the layout of a shortened variant of the + ZipIt extra block for Macintosh used only for directory + entries. This variant is used by ZipIt 1.3.5 and newer to + save some optional Mac-specific information about directories. + The local-header and central-header versions are identical. + + Value Size Description + ----- ---- ----------- + (Mac2c) 0x2805 Short tag for this extra block type + TSize Short total data size for this block (12) + "ZPIT" beLong extra-field signature + frFlags beShort attributes from DInfo.frFlags, MAY + be omitted + View beShort ZipIt view flag, MAY be omitted + + + The View field specifies ZipIt-internal settings as follows: + + Bits of the Flags: + bit 0 if set, the folder is shown expanded (open) + when the archive contents are viewed in ZipIt. + bits 1-15 reserved, zero; + + + 4.6.7 -FWKCS MD5 Extra Field (0x4b46): + + The FWKCS Contents_Signature System, used in + automatically identifying files independent of file name, + optionally adds and uses an extra field to support the + rapid creation of an enhanced contents_signature: + + Header ID = 0x4b46 + Data Size = 0x0013 + Preface = 'M','D','5' + followed by 16 bytes containing the uncompressed file's + 128_bit MD5 hash(1), low byte first. + + When FWKCS revises a .ZIP file central directory to add + this extra field for a file, it also replaces the + central directory entry for that file's uncompressed + file length with a measured value. + + FWKCS provides an option to strip this extra field, if + present, from a .ZIP file central directory. In adding + this extra field, FWKCS preserves .ZIP file Authenticity + Verification; if stripping this extra field, FWKCS + preserves all versions of AV through PKZIP version 2.04g. + + FWKCS, and FWKCS Contents_Signature System, are + trademarks of Frederick W. Kantor. + + (1) R. Rivest, RFC1321.TXT, MIT Laboratory for Computer + Science and RSA Data Security, Inc., April 1992. + ll.76-77: "The MD5 algorithm is being placed in the + public domain for review and possible adoption as a + standard." + + + 4.6.8 -Info-ZIP Unicode Comment Extra Field (0x6375): + + Stores the UTF-8 version of the file comment as stored in the + central directory header. (Last Revision 20070912) + + Value Size Description + ----- ---- ----------- + (UCom) 0x6375 Short tag for this extra block type ("uc") + TSize Short total data size for this block + Version 1 byte version of this extra field, currently 1 + ComCRC32 4 bytes Comment Field CRC32 Checksum + UnicodeCom Variable UTF-8 version of the entry comment + + Currently Version is set to the number 1. If there is a need + to change this field, the version will be incremented. Changes + MAY NOT be backward compatible so this extra field SHOULD NOT be + used if the version is not recognized. + + The ComCRC32 is the standard zip CRC32 checksum of the File Comment + field in the central directory header. This is used to verify that + the comment field has not changed since the Unicode Comment extra field + was created. This can happen if a utility changes the File Comment + field but does not update the UTF-8 Comment extra field. If the CRC + check fails, this Unicode Comment extra field SHOULD be ignored and + the File Comment field in the header SHOULD be used instead. + + The UnicodeCom field is the UTF-8 version of the File Comment field + in the header. As UnicodeCom is defined to be UTF-8, no UTF-8 byte + order mark (BOM) is used. The length of this field is determined by + subtracting the size of the previous fields from TSize. If both the + File Name and Comment fields are UTF-8, the new General Purpose Bit + Flag, bit 11 (Language encoding flag (EFS)), can be used to indicate + both the header File Name and Comment fields are UTF-8 and, in this + case, the Unicode Path and Unicode Comment extra fields are not + needed and SHOULD NOT be created. Note that, for backward + compatibility, bit 11 SHOULD only be used if the native character set + of the paths and comments being zipped up are already in UTF-8. It is + expected that the same file comment storage method, either general + purpose bit 11 or extra fields, be used in both the Local and Central + Directory Header for a file. + + + 4.6.9 -Info-ZIP Unicode Path Extra Field (0x7075): + + Stores the UTF-8 version of the file name field as stored in the + local header and central directory header. (Last Revision 20070912) + + Value Size Description + ----- ---- ----------- + (UPath) 0x7075 Short tag for this extra block type ("up") + TSize Short total data size for this block + Version 1 byte version of this extra field, currently 1 + NameCRC32 4 bytes File Name Field CRC32 Checksum + UnicodeName Variable UTF-8 version of the entry File Name + + Currently Version is set to the number 1. If there is a need + to change this field, the version will be incremented. Changes + MAY NOT be backward compatible so this extra field SHOULD NOT be + used if the version is not recognized. + + The NameCRC32 is the standard zip CRC32 checksum of the File Name + field in the header. This is used to verify that the header + File Name field has not changed since the Unicode Path extra field + was created. This can happen if a utility renames the File Name but + does not update the UTF-8 path extra field. If the CRC check fails, + this UTF-8 Path Extra Field SHOULD be ignored and the File Name field + in the header SHOULD be used instead. + + The UnicodeName is the UTF-8 version of the contents of the File Name + field in the header. As UnicodeName is defined to be UTF-8, no UTF-8 + byte order mark (BOM) is used. The length of this field is determined + by subtracting the size of the previous fields from TSize. If both + the File Name and Comment fields are UTF-8, the new General Purpose + Bit Flag, bit 11 (Language encoding flag (EFS)), can be used to + indicate that both the header File Name and Comment fields are UTF-8 + and, in this case, the Unicode Path and Unicode Comment extra fields + are not needed and SHOULD NOT be created. Note that, for backward + compatibility, bit 11 SHOULD only be used if the native character set + of the paths and comments being zipped up are already in UTF-8. It is + expected that the same file name storage method, either general + purpose bit 11 or extra fields, be used in both the Local and Central + Directory Header for a file. + + + 4.6.10 -Microsoft Open Packaging Growth Hint (0xa220): + + Value Size Description + ----- ---- ----------- + 0xa220 Short tag for this extra block type + TSize Short size of Sig + PadVal + Padding + Sig Short verification signature (A028) + PadVal Short Initial padding value + Padding variable filled with NULL characters + + 4.6.11 -Data Stream Alignment (Apache Commons-Compress) (0xa11e): + + (per Zbynek Vyskovsky) Defines alignment of data stream of this + entry within the zip archive. Additionally, indicates whether the + compression method should be kept when re-compressing the zip file. + + The purpose of this extra field is to align specific resources to + word or page boundaries so they can be easily mapped into memory. + + Value Size Description + ----- ---- ----------- + 0xa11e Short tag for this extra block type + TSize Short total data size for this block (2+padding) + alignment Short required alignment and indicator + 0x00 Variable padding + + The alignment field (lower 15 bits) defines the minimal alignment + required by the data stream. Bit 15 of alignment field indicates + whether the compression method of this entry can be changed when + recompressing the zip file. The value 0 means the compression method + should not be changed. The value 1 indicates the compression method + may be changed. The padding field contains padding to ensure the correct + alignment. It can be changed at any time when the offset or required + alignment changes. (see https://issues.apache.org/jira/browse/COMPRESS-391) + + +4.7 Manifest Files +------------------ + + 4.7.1 Applications using ZIP files MAY have a need for additional + information that MUST be included with the files placed into + a ZIP file. Application specific information that cannot be + stored using the defined ZIP storage records SHOULD be stored + using the extensible Extra Field convention defined in this + document. However, some applications MAY use a manifest + file as a means for storing additional information. One + example is the META-INF/MANIFEST.MF file used in ZIP formatted + files having the .JAR extension (JAR files). + + 4.7.2 A manifest file is a file created for the application process + that requires this information. A manifest file MAY be of any + file type required by the defining application process. It is + placed within the same ZIP file as files to which this information + applies. By convention, this file is typically the first file placed + into the ZIP file and it MAY include a defined directory path. + + 4.7.3 Manifest files MAY be compressed or encrypted as needed for + application processing of the files inside the ZIP files. + + Manifest files are outside of the scope of this specification. + + +5.0 Explanation of compression methods +-------------------------------------- + + +5.1 UnShrinking - Method 1 +-------------------------- + + 5.1.1 Shrinking is a Dynamic Ziv-Lempel-Welch compression algorithm + with partial clearing. The initial code size is 9 bits, and the + maximum code size is 13 bits. Shrinking differs from conventional + Dynamic Ziv-Lempel-Welch implementations in several respects: + + 5.1.2 The code size is controlled by the compressor, and is + not automatically increased when codes larger than the current + code size are created (but not necessarily used). When + the decompressor encounters the code sequence 256 + (decimal) followed by 1, it SHOULD increase the code size + read from the input stream to the next bit size. No + blocking of the codes is performed, so the next code at + the increased size SHOULD be read from the input stream + immediately after where the previous code at the smaller + bit size was read. Again, the decompressor SHOULD NOT + increase the code size used until the sequence 256,1 is + encountered. + + 5.1.3 When the table becomes full, total clearing is not + performed. Rather, when the compressor emits the code + sequence 256,2 (decimal), the decompressor SHOULD clear + all leaf nodes from the Ziv-Lempel tree, and continue to + use the current code size. The nodes that are cleared + from the Ziv-Lempel tree are then re-used, with the lowest + code value re-used first, and the highest code value + re-used last. The compressor can emit the sequence 256,2 + at any time. + +5.2 Expanding - Methods 2-5 +--------------------------- + + 5.2.1 The Reducing algorithm is actually a combination of two + distinct algorithms. The first algorithm compresses repeated + byte sequences, and the second algorithm takes the compressed + stream from the first algorithm and applies a probabilistic + compression method. + + 5.2.2 The probabilistic compression stores an array of 'follower + sets' S(j), for j=0 to 255, corresponding to each possible + ASCII character. Each set contains between 0 and 32 + characters, to be denoted as S(j)[0],...,S(j)[m], where m<32. + The sets are stored at the beginning of the data area for a + Reduced file, in reverse order, with S(255) first, and S(0) + last. + + 5.2.3 The sets are encoded as { N(j), S(j)[0],...,S(j)[N(j)-1] }, + where N(j) is the size of set S(j). N(j) can be 0, in which + case the follower set for S(j) is empty. Each N(j) value is + encoded in 6 bits, followed by N(j) eight bit character values + corresponding to S(j)[0] to S(j)[N(j)-1] respectively. If + N(j) is 0, then no values for S(j) are stored, and the value + for N(j-1) immediately follows. + + 5.2.4 Immediately after the follower sets, is the compressed data + stream. The compressed data stream can be interpreted for the + probabilistic decompression as follows: + + let Last-Character <- 0. + loop until done + if the follower set S(Last-Character) is empty then + read 8 bits from the input stream, and copy this + value to the output stream. + otherwise if the follower set S(Last-Character) is non-empty then + read 1 bit from the input stream. + if this bit is not zero then + read 8 bits from the input stream, and copy this + value to the output stream. + otherwise if this bit is zero then + read B(N(Last-Character)) bits from the input + stream, and assign this value to I. + Copy the value of S(Last-Character)[I] to the + output stream. + + assign the last value placed on the output stream to + Last-Character. + end loop + + B(N(j)) is defined as the minimal number of bits required to + encode the value N(j)-1. + + 5.2.5 The decompressed stream from above can then be expanded to + re-create the original file as follows: + + let State <- 0. + + loop until done + read 8 bits from the input stream into C. + case State of + 0: if C is not equal to DLE (144 decimal) then + copy C to the output stream. + otherwise if C is equal to DLE then + let State <- 1. + + 1: if C is non-zero then + let V <- C. + let Len <- L(V) + let State <- F(Len). + otherwise if C is zero then + copy the value 144 (decimal) to the output stream. + let State <- 0 + + 2: let Len <- Len + C + let State <- 3. + + 3: move backwards D(V,C) bytes in the output stream + (if this position is before the start of the output + stream, then assume that all the data before the + start of the output stream is filled with zeros). + copy Len+3 bytes from this position to the output stream. + let State <- 0. + end case + end loop + + The functions F,L, and D are dependent on the 'compression + factor', 1 through 4, and are defined as follows: + + For compression factor 1: + L(X) equals the lower 7 bits of X. + F(X) equals 2 if X equals 127 otherwise F(X) equals 3. + D(X,Y) equals the (upper 1 bit of X) * 256 + Y + 1. + For compression factor 2: + L(X) equals the lower 6 bits of X. + F(X) equals 2 if X equals 63 otherwise F(X) equals 3. + D(X,Y) equals the (upper 2 bits of X) * 256 + Y + 1. + For compression factor 3: + L(X) equals the lower 5 bits of X. + F(X) equals 2 if X equals 31 otherwise F(X) equals 3. + D(X,Y) equals the (upper 3 bits of X) * 256 + Y + 1. + For compression factor 4: + L(X) equals the lower 4 bits of X. + F(X) equals 2 if X equals 15 otherwise F(X) equals 3. + D(X,Y) equals the (upper 4 bits of X) * 256 + Y + 1. + +5.3 Imploding - Method 6 +------------------------ + + 5.3.1 The Imploding algorithm is actually a combination of two + distinct algorithms. The first algorithm compresses repeated byte + sequences using a sliding dictionary. The second algorithm is + used to compress the encoding of the sliding dictionary output, + using multiple Shannon-Fano trees. + + 5.3.2 The Imploding algorithm can use a 4K or 8K sliding dictionary + size. The dictionary size used can be determined by bit 1 in the + general purpose flag word; a 0 bit indicates a 4K dictionary + while a 1 bit indicates an 8K dictionary. + + 5.3.3 The Shannon-Fano trees are stored at the start of the + compressed file. The number of trees stored is defined by bit 2 in + the general purpose flag word; a 0 bit indicates two trees stored, + a 1 bit indicates three trees are stored. If 3 trees are stored, + the first Shannon-Fano tree represents the encoding of the + Literal characters, the second tree represents the encoding of + the Length information, the third represents the encoding of the + Distance information. When 2 Shannon-Fano trees are stored, the + Length tree is stored first, followed by the Distance tree. + + 5.3.4 The Literal Shannon-Fano tree, if present is used to represent + the entire ASCII character set, and contains 256 values. This + tree is used to compress any data not compressed by the sliding + dictionary algorithm. When this tree is present, the Minimum + Match Length for the sliding dictionary is 3. If this tree is + not present, the Minimum Match Length is 2. + + 5.3.5 The Length Shannon-Fano tree is used to compress the Length + part of the (length,distance) pairs from the sliding dictionary + output. The Length tree contains 64 values, ranging from the + Minimum Match Length, to 63 plus the Minimum Match Length. + + 5.3.6 The Distance Shannon-Fano tree is used to compress the Distance + part of the (length,distance) pairs from the sliding dictionary + output. The Distance tree contains 64 values, ranging from 0 to + 63, representing the upper 6 bits of the distance value. The + distance values themselves will be between 0 and the sliding + dictionary size, either 4K or 8K. + + 5.3.7 The Shannon-Fano trees themselves are stored in a compressed + format. The first byte of the tree data represents the number of + bytes of data representing the (compressed) Shannon-Fano tree + minus 1. The remaining bytes represent the Shannon-Fano tree + data encoded as: + + High 4 bits: Number of values at this bit length + 1. (1 - 16) + Low 4 bits: Bit Length needed to represent value + 1. (1 - 16) + + 5.3.8 The Shannon-Fano codes can be constructed from the bit lengths + using the following algorithm: + + 1) Sort the Bit Lengths in ascending order, while retaining the + order of the original lengths stored in the file. + + 2) Generate the Shannon-Fano trees: + + Code <- 0 + CodeIncrement <- 0 + LastBitLength <- 0 + i <- number of Shannon-Fano codes - 1 (either 255 or 63) + + loop while i >= 0 + Code = Code + CodeIncrement + if BitLength(i) <> LastBitLength then + LastBitLength=BitLength(i) + CodeIncrement = 1 shifted left (16 - LastBitLength) + ShannonCode(i) = Code + i <- i - 1 + end loop + + 3) Reverse the order of all the bits in the above ShannonCode() + vector, so that the most significant bit becomes the least + significant bit. For example, the value 0x1234 (hex) would + become 0x2C48 (hex). + + 4) Restore the order of Shannon-Fano codes as originally stored + within the file. + + Example: + + This example will show the encoding of a Shannon-Fano tree + of size 8. Notice that the actual Shannon-Fano trees used + for Imploding are either 64 or 256 entries in size. + + Example: 0x02, 0x42, 0x01, 0x13 + + The first byte indicates 3 values in this table. Decoding the + bytes: + 0x42 = 5 codes of 3 bits long + 0x01 = 1 code of 2 bits long + 0x13 = 2 codes of 4 bits long + + This would generate the original bit length array of: + (3, 3, 3, 3, 3, 2, 4, 4) + + There are 8 codes in this table for the values 0 thru 7. Using + the algorithm to obtain the Shannon-Fano codes produces: + + Reversed Order Original + Val Sorted Constructed Code Value Restored Length + --- ------ ----------------- -------- -------- ------ + 0: 2 1100000000000000 11 101 3 + 1: 3 1010000000000000 101 001 3 + 2: 3 1000000000000000 001 110 3 + 3: 3 0110000000000000 110 010 3 + 4: 3 0100000000000000 010 100 3 + 5: 3 0010000000000000 100 11 2 + 6: 4 0001000000000000 1000 1000 4 + 7: 4 0000000000000000 0000 0000 4 + + The values in the Val, Order Restored and Original Length columns + now represent the Shannon-Fano encoding tree that can be used for + decoding the Shannon-Fano encoded data. How to parse the + variable length Shannon-Fano values from the data stream is beyond + the scope of this document. (See the references listed at the end of + this document for more information.) However, traditional decoding + schemes used for Huffman variable length decoding, such as the + Greenlaw algorithm, can be successfully applied. + + 5.3.9 The compressed data stream begins immediately after the + compressed Shannon-Fano data. The compressed data stream can be + interpreted as follows: + + loop until done + read 1 bit from input stream. + + if this bit is non-zero then (encoded data is literal data) + if Literal Shannon-Fano tree is present + read and decode character using Literal Shannon-Fano tree. + otherwise + read 8 bits from input stream. + copy character to the output stream. + otherwise (encoded data is sliding dictionary match) + if 8K dictionary size + read 7 bits for offset Distance (lower 7 bits of offset). + otherwise + read 6 bits for offset Distance (lower 6 bits of offset). + + using the Distance Shannon-Fano tree, read and decode the + upper 6 bits of the Distance value. + + using the Length Shannon-Fano tree, read and decode + the Length value. + + Length <- Length + Minimum Match Length + + if Length = 63 + Minimum Match Length + read 8 bits from the input stream, + add this value to Length. + + move backwards Distance+1 bytes in the output stream, and + copy Length characters from this position to the output + stream. (if this position is before the start of the output + stream, then assume that all the data before the start of + the output stream is filled with zeros). + end loop + +5.4 Tokenizing - Method 7 +------------------------- + + 5.4.1 This method is not used by PKZIP. + +5.5 Deflating - Method 8 +------------------------ + + 5.5.1 The Deflate algorithm is similar to the Implode algorithm using + a sliding dictionary of up to 32K with secondary compression + from Huffman/Shannon-Fano codes. + + 5.5.2 The compressed data is stored in blocks with a header describing + the block and the Huffman codes used in the data block. The header + format is as follows: + + Bit 0: Last Block bit This bit is set to 1 if this is the last + compressed block in the data. + Bits 1-2: Block type + 00 (0) - Block is stored - All stored data is byte aligned. + Skip bits until next byte, then next word = block + length, followed by the ones compliment of the block + length word. Remaining data in block is the stored + data. + + 01 (1) - Use fixed Huffman codes for literal and distance codes. + Lit Code Bits Dist Code Bits + --------- ---- --------- ---- + 0 - 143 8 0 - 31 5 + 144 - 255 9 + 256 - 279 7 + 280 - 287 8 + + Literal codes 286-287 and distance codes 30-31 are + never used but participate in the huffman construction. + + 10 (2) - Dynamic Huffman codes. (See expanding Huffman codes) + + 11 (3) - Reserved - Flag a "Error in compressed data" if seen. + + 5.5.3 Expanding Huffman Codes + + If the data block is stored with dynamic Huffman codes, the Huffman + codes are sent in the following compressed format: + + 5 Bits: # of Literal codes sent - 256 (256 - 286) + All other codes are never sent. + 5 Bits: # of Dist codes - 1 (1 - 32) + 4 Bits: # of Bit Length codes - 3 (3 - 19) + + The Huffman codes are sent as bit lengths and the codes are built as + described in the implode algorithm. The bit lengths themselves are + compressed with Huffman codes. There are 19 bit length codes: + + 0 - 15: Represent bit lengths of 0 - 15 + 16: Copy the previous bit length 3 - 6 times. + The next 2 bits indicate repeat length (0 = 3, ... ,3 = 6) + Example: Codes 8, 16 (+2 bits 11), 16 (+2 bits 10) will + expand to 12 bit lengths of 8 (1 + 6 + 5) + 17: Repeat a bit length of 0 for 3 - 10 times. (3 bits of length) + 18: Repeat a bit length of 0 for 11 - 138 times (7 bits of length) + + The lengths of the bit length codes are sent packed 3 bits per value + (0 - 7) in the following order: + + 16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2, 14, 1, 15 + + The Huffman codes SHOULD be built as described in the Implode algorithm + except codes are assigned starting at the shortest bit length, i.e. the + shortest code SHOULD be all 0's rather than all 1's. Also, codes with + a bit length of zero do not participate in the tree construction. The + codes are then used to decode the bit lengths for the literal and + distance tables. + + The bit lengths for the literal tables are sent first with the number + of entries sent described by the 5 bits sent earlier. There are up + to 286 literal characters; the first 256 represent the respective 8 + bit character, code 256 represents the End-Of-Block code, the remaining + 29 codes represent copy lengths of 3 thru 258. There are up to 30 + distance codes representing distances from 1 thru 32k as described + below. + + Length Codes + ------------ + Extra Extra Extra Extra + Code Bits Length Code Bits Lengths Code Bits Lengths Code Bits Length(s) + ---- ---- ------ ---- ---- ------- ---- ---- ------- ---- ---- --------- + 257 0 3 265 1 11,12 273 3 35-42 281 5 131-162 + 258 0 4 266 1 13,14 274 3 43-50 282 5 163-194 + 259 0 5 267 1 15,16 275 3 51-58 283 5 195-226 + 260 0 6 268 1 17,18 276 3 59-66 284 5 227-257 + 261 0 7 269 2 19-22 277 4 67-82 285 0 258 + 262 0 8 270 2 23-26 278 4 83-98 + 263 0 9 271 2 27-30 279 4 99-114 + 264 0 10 272 2 31-34 280 4 115-130 + + Distance Codes + -------------- + Extra Extra Extra Extra + Code Bits Dist Code Bits Dist Code Bits Distance Code Bits Distance + ---- ---- ---- ---- ---- ------ ---- ---- -------- ---- ---- -------- + 0 0 1 8 3 17-24 16 7 257-384 24 11 4097-6144 + 1 0 2 9 3 25-32 17 7 385-512 25 11 6145-8192 + 2 0 3 10 4 33-48 18 8 513-768 26 12 8193-12288 + 3 0 4 11 4 49-64 19 8 769-1024 27 12 12289-16384 + 4 1 5,6 12 5 65-96 20 9 1025-1536 28 13 16385-24576 + 5 1 7,8 13 5 97-128 21 9 1537-2048 29 13 24577-32768 + 6 2 9-12 14 6 129-192 22 10 2049-3072 + 7 2 13-16 15 6 193-256 23 10 3073-4096 + + 5.5.4 The compressed data stream begins immediately after the + compressed header data. The compressed data stream can be + interpreted as follows: + + do + read header from input stream. + + if stored block + skip bits until byte aligned + read count and 1's compliment of count + copy count bytes data block + otherwise + loop until end of block code sent + decode literal character from input stream + if literal < 256 + copy character to the output stream + otherwise + if literal = end of block + break from loop + otherwise + decode distance from input stream + + move backwards distance bytes in the output stream, and + copy length characters from this position to the output + stream. + end loop + while not last block + + if data descriptor exists + skip bits until byte aligned + read crc and sizes + endif + +5.6 Enhanced Deflating - Method 9 +--------------------------------- + + 5.6.1 The Enhanced Deflating algorithm is similar to Deflate but uses + a sliding dictionary of up to 64K. Deflate64(tm) is supported + by the Deflate extractor. + +5.7 BZIP2 - Method 12 +--------------------- + + 5.7.1 BZIP2 is an open-source data compression algorithm developed by + Julian Seward. Information and source code for this algorithm + can be found on the internet. + +5.8 LZMA - Method 14 +--------------------- + + 5.8.1 LZMA is a block-oriented, general purpose data compression + algorithm developed and maintained by Igor Pavlov. It is a derivative + of LZ77 that utilizes Markov chains and a range coder. Information and + source code for this algorithm can be found on the internet. Consult + with the author of this algorithm for information on terms or + restrictions on use. + + Support for LZMA within the ZIP format is defined as follows: + + 5.8.2 The Compression method field within the ZIP Local and Central + Header records will be set to the value 14 to indicate data was + compressed using LZMA. + + 5.8.3 The Version needed to extract field within the ZIP Local and + Central Header records will be set to 6.3 to indicate the minimum + ZIP format version supporting this feature. + + 5.8.4 File data compressed using the LZMA algorithm MUST be placed + immediately following the Local Header for the file. If a standard + ZIP encryption header is required, it will follow the Local Header + and will precede the LZMA compressed file data segment. The location + of LZMA compressed data segment within the ZIP format will be as shown: + + [local header file 1] + [encryption header file 1] + [LZMA compressed data segment for file 1] + [data descriptor 1] + [local header file 2] + + 5.8.5 The encryption header and data descriptor records MAY + be conditionally present. The LZMA Compressed Data Segment + will consist of an LZMA Properties Header followed by the + LZMA Compressed Data as shown: + + [LZMA properties header for file 1] + [LZMA compressed data for file 1] + + 5.8.6 The LZMA Compressed Data will be stored as provided by the + LZMA compression library. Compressed size, uncompressed size and + other file characteristics about the file being compressed MUST be + stored in standard ZIP storage format. + + 5.8.7 The LZMA Properties Header will store specific data required + to decompress the LZMA compressed Data. This data is set by the + LZMA compression engine using the function WriteCoderProperties() + as documented within the LZMA SDK. + + 5.8.8 Storage fields for the property information within the LZMA + Properties Header are as follows: + + LZMA Version Information 2 bytes + LZMA Properties Size 2 bytes + LZMA Properties Data variable, defined by "LZMA Properties Size" + + 5.8.8.1 LZMA Version Information - this field identifies which version + of the LZMA SDK was used to compress a file. The first byte will + store the major version number of the LZMA SDK and the second + byte will store the minor number. + + 5.8.8.2 LZMA Properties Size - this field defines the size of the + remaining property data. Typically this size SHOULD be determined by + the version of the SDK. This size field is included as a convenience + and to help avoid any ambiguity arising in the future due + to changes in this compression algorithm. + + 5.8.8.3 LZMA Property Data - this variable sized field records the + required values for the decompressor as defined by the LZMA SDK. + The data stored in this field SHOULD be obtained using the + WriteCoderProperties() in the version of the SDK defined by + the "LZMA Version Information" field. + + 5.8.8.4 The layout of the "LZMA Properties Data" field is a function of + the LZMA compression algorithm. It is possible that this layout MAY be + changed by the author over time. The data layout in version 4.3 of the + LZMA SDK defines a 5 byte array that uses 4 bytes to store the dictionary + size in little-endian order. This is preceded by a single packed byte as + the first element of the array that contains the following fields: + + PosStateBits + LiteralPosStateBits + LiteralContextBits + + Refer to the LZMA documentation for a more detailed explanation of + these fields. + + 5.8.9 Data compressed with method 14, LZMA, MAY include an end-of-stream + (EOS) marker ending the compressed data stream. This marker is not + required, but its use is highly recommended to facilitate processing + and implementers SHOULD include the EOS marker whenever possible. + When the EOS marker is used, general purpose bit 1 MUSY be set. If + general purpose bit 1 is not set, the EOS marker is not present. + +5.9 WavPack - Method 97 +----------------------- + + 5.9.1 Information describing the use of compression method 97 is + provided by WinZIP International, LLC. This method relies on the + open source WavPack audio compression utility developed by David Bryant. + Information on WavPack is available at www.wavpack.com. Please consult + with the author of this algorithm for information on terms and + restrictions on use. + + 5.9.2 WavPack data for a file begins immediately after the end of the + local header data. This data is the output from WavPack compression + routines. Within the ZIP file, the use of WavPack compression is + indicated by setting the compression method field to a value of 97 + in both the local header and the central directory header. The Version + needed to extract and version made by fields use the same values as are + used for data compressed using the Deflate algorithm. + + 5.9.3 An implementation note for storing digital sample data when using + WavPack compression within ZIP files is that all of the bytes of + the sample data SHOULD be compressed. This includes any unused + bits up to the byte boundary. An example is a 2 byte sample that + uses only 12 bits for the sample data with 4 unused bits. If only + 12 bits are passed as the sample size to the WavPack routines, the 4 + unused bits will be set to 0 on extraction regardless of their original + state. To avoid this, the full 16 bits of the sample data size + SHOULD be provided. + +5.10 PPMd - Method 98 +--------------------- + + 5.10.1 PPMd is a data compression algorithm developed by Dmitry Shkarin + which includes a carryless rangecoder developed by Dmitry Subbotin. + This algorithm is based on predictive phrase matching on multiple + order contexts. Information and source code for this algorithm + can be found on the internet. Consult with the author of this + algorithm for information on terms or restrictions on use. + + 5.10.2 Support for PPMd within the ZIP format currently is provided only + for version I, revision 1 of the algorithm. Storage requirements + for using this algorithm are as follows: + + 5.10.3 Parameters needed to control the algorithm are stored in the two + bytes immediately preceding the compressed data. These bytes are + used to store the following fields: + + Model order - sets the maximum model order, default is 8, possible + values are from 2 to 16 inclusive + + Sub-allocator size - sets the size of sub-allocator in MB, default is 50, + possible values are from 1MB to 256MB inclusive + + Model restoration method - sets the method used to restart context + model at memory insufficiency, values are: + + 0 - restarts model from scratch - default + 1 - cut off model - decreases performance by as much as 2x + 2 - freeze context tree - not recommended + + 5.10.4 An example for packing these fields into the 2 byte storage field is + illustrated below. These values are stored in Intel low-byte/high-byte + order. + + wPPMd = (Model order - 1) + + ((Sub-allocator size - 1) << 4) + + (Model restoration method << 12) + + +5.11 AE-x Encryption marker - Method 99 +------------------------------------------- + +5.12 JPEG variant - Method 96 +------------------------------------------- + +5.13 PKWARE Data Compression Library Imploding - Method 10 +----------------------------------------------------------- + +5.14 Reserved - Method 11 +------------------------------------------- + +5.15 Reserved - Method 13 +------------------------------------------- + +5.16 Reserved - Method 15 +------------------------------------------- + +5.17 IBM z/OS CMPSC Compression - Method 16 +------------------------------------------- + +Method 16 utilizes the IBM hardware compression facility available +on most IBM mainframes. Hardware compression can significantly +increase the speed of data compression. This method uses a variant +of the LZ78 algorithm. CMPSC hardware compression is performed +using the COMPRESSION CALL instruction. + +ZIP archives can be created using this method only on mainframes +supporting the CP instruction. Extraction MAY occur on any +platform supporting this compression algorithm. Use of this +algorithm requires creation of a compression dictionary and +an expansion dictionary. The expansion dictionary MUST be +placed into the ZIP archive for use on the system where +extraction will occur. + +Additional information on this compression algorithm and dictionaries +can be found in the IBM provided document titled IBM ESA/390 Data +Compression (SA22-7208-01). Storage requirements for using CMPSC +compression are as follows. + +The format for the compressed data stream placed into the ZIP +archive following the Local Header is: + + [dictionary header] + [expansion dictionary] + [CMPSC compressed data] + +If encryption is used to encrypt a file compressed with CMPSC, these +sections MUST be encrypted as a single entity. + +The format of the dictionary header is: + + Value Size Description + ----- ---- ----------- + Version 1 byte 1 + Flags/Symsize 1 byte Processing flags and + symbol size + DictionaryLen 4 bytes Length of the + expansion dictionary + +Explanation of processing flags and symbol size: + +The high 4 bits are used to store the processing flags. The low +4 bits represent the size of a symbol, in bits (values range +from 9-13). Flag values are defined below. + + 0x80 - expansion dictionary + 0x40 - expansion dictionary is compressed using Deflate + 0x20 - Reserved + 0x10 - Reserved + + +5.18 Reserved - Method 17 +------------------------------------------- + +5.19 IBM TERSE - Method 18 +------------------------------------------- + +5.20 IBM LZ77 z Architecture - Method 19 +----------------------------------------- + +6.0 Traditional PKWARE Encryption +---------------------------------- + + 6.0.1 The following information discusses the decryption steps + required to support traditional PKWARE encryption. This + form of encryption is considered weak by today's standards + and its use is recommended only for situations with + low security needs or for compatibility with older .ZIP + applications. + +6.1 Traditional PKWARE Decryption +--------------------------------- + + 6.1.1 PKWARE is grateful to Mr. Roger Schlafly for his expert + contribution towards the development of PKWARE's traditional + encryption. + + 6.1.2 PKZIP encrypts the compressed data stream. Encrypted files + MUST be decrypted before they can be extracted to their original + form. + + 6.1.3 Each encrypted file has an extra 12 bytes stored at the start + of the data area defining the encryption header for that file. The + encryption header is originally set to random values, and then + itself encrypted, using three, 32-bit keys. The key values are + initialized using the supplied encryption password. After each byte + is encrypted, the keys are then updated using pseudo-random number + generation techniques in combination with the same CRC-32 algorithm + used in PKZIP and described elsewhere in this document. + + 6.1.4 The following are the basic steps required to decrypt a file: + + 1) Initialize the three 32-bit keys with the password. + 2) Read and decrypt the 12-byte encryption header, further + initializing the encryption keys. + 3) Read and decrypt the compressed data stream using the + encryption keys. + + 6.1.5 Initializing the encryption keys + + Key(0) <- 305419896 + Key(1) <- 591751049 + Key(2) <- 878082192 + + loop for i <- 0 to length(password)-1 + update_keys(password(i)) + end loop + + Where update_keys() is defined as: + + update_keys(char): + Key(0) <- crc32(key(0),char) + Key(1) <- Key(1) + (Key(0) & 000000ffH) + Key(1) <- Key(1) * 134775813 + 1 + Key(2) <- crc32(key(2),key(1) >> 24) + end update_keys + + Where crc32(old_crc,char) is a routine that given a CRC value and a + character, returns an updated CRC value after applying the CRC-32 + algorithm described elsewhere in this document. + + 6.1.6 Decrypting the encryption header + + The purpose of this step is to further initialize the encryption + keys, based on random data, to render a plaintext attack on the + data ineffective. + + Read the 12-byte encryption header into Buffer, in locations + Buffer(0) thru Buffer(11). + + loop for i <- 0 to 11 + C <- buffer(i) ^ decrypt_byte() + update_keys(C) + buffer(i) <- C + end loop + + Where decrypt_byte() is defined as: + + unsigned char decrypt_byte() + local unsigned short temp + temp <- Key(2) | 2 + decrypt_byte <- (temp * (temp ^ 1)) >> 8 + end decrypt_byte + + After the header is decrypted, the last 1 or 2 bytes in Buffer + SHOULD be the high-order word/byte of the CRC for the file being + decrypted, stored in Intel low-byte/high-byte order. Versions of + PKZIP prior to 2.0 used a 2 byte CRC check; a 1 byte CRC check is + used on versions after 2.0. This can be used to test if the password + supplied is correct or not. + + 6.1.7 Decrypting the compressed data stream + + The compressed data stream can be decrypted as follows: + + loop until done + read a character into C + Temp <- C ^ decrypt_byte() + update_keys(temp) + output Temp + end loop + + +7.0 Strong Encryption Specification +----------------------------------- + + 7.0.1 Portions of the Strong Encryption technology defined in this + specification are covered under patents and pending patent applications. + Refer to the section in this document entitled "Incorporating + PKWARE Proprietary Technology into Your Product" for more information. + +7.1 Strong Encryption Overview +------------------------------ + + 7.1.1 Version 5.x of this specification introduced support for strong + encryption algorithms. These algorithms can be used with either + a password or an X.509v3 digital certificate to encrypt each file. + This format specification supports either password or certificate + based encryption to meet the security needs of today, to enable + interoperability between users within both PKI and non-PKI + environments, and to ensure interoperability between different + computing platforms that are running a ZIP program. + + 7.1.2 Password based encryption is the most common form of encryption + people are familiar with. However, inherent weaknesses with + passwords (e.g. susceptibility to dictionary/brute force attack) + as well as password management and support issues make certificate + based encryption a more secure and scalable option. Industry + efforts and support are defining and moving towards more advanced + security solutions built around X.509v3 digital certificates and + Public Key Infrastructures(PKI) because of the greater scalability, + administrative options, and more robust security over traditional + password based encryption. + + 7.1.3 Most standard encryption algorithms are supported with this + specification. Reference implementations for many of these + algorithms are available from either commercial or open source + distributors. Readily available cryptographic toolkits make + implementation of the encryption features straight-forward. + This document is not intended to provide a treatise on data + encryption principles or theory. Its purpose is to document the + data structures required for implementing interoperable data + encryption within the .ZIP format. It is strongly recommended that + you have a good understanding of data encryption before reading + further. + + 7.1.4 The algorithms introduced in Version 5.0 of this specification + include: + + RC2 40 bit, 64 bit, and 128 bit + RC4 40 bit, 64 bit, and 128 bit + DES + 3DES 112 bit and 168 bit + + Version 5.1 adds support for the following: + + AES 128 bit, 192 bit, and 256 bit + + + 7.1.5 Version 6.1 introduces encryption data changes to support + interoperability with Smartcard and USB Token certificate storage + methods which do not support the OAEP strengthening standard. + + 7.1.6 Version 6.2 introduces support for encrypting metadata by compressing + and encrypting the central directory data structure to reduce information + leakage. Information leakage can occur in legacy ZIP applications + through exposure of information about a file even though that file is + stored encrypted. The information exposed consists of file + characteristics stored within the records and fields defined by this + specification. This includes data such as a file's name, its original + size, timestamp and CRC32 value. + + 7.1.7 Version 6.3 introduces support for encrypting data using the Blowfish + and Twofish algorithms. These are symmetric block ciphers developed + by Bruce Schneier. Blowfish supports using a variable length key from + 32 to 448 bits. Block size is 64 bits. Implementations SHOULD use 16 + rounds and the only mode supported within ZIP files is CBC. Twofish + supports key sizes 128, 192 and 256 bits. Block size is 128 bits. + Implementations SHOULD use 16 rounds and the only mode supported within + ZIP files is CBC. Information and source code for both Blowfish and + Twofish algorithms can be found on the internet. Consult with the author + of these algorithms for information on terms or restrictions on use. + + 7.1.8 Central Directory Encryption provides greater protection against + information leakage by encrypting the Central Directory structure and + by masking key values that are replicated in the unencrypted Local + Header. ZIP compatible programs that cannot interpret an encrypted + Central Directory structure cannot rely on the data in the corresponding + Local Header for decompression information. + + 7.1.9 Extra Field records that MAY contain information about a file that SHOULD + not be exposed SHOULD NOT be stored in the Local Header and SHOULD only + be written to the Central Directory where they can be encrypted. This + design currently does not support streaming. Information in the End of + Central Directory record, the Zip64 End of Central Directory Locator, + and the Zip64 End of Central Directory records are not encrypted. Access + to view data on files within a ZIP file with an encrypted Central Directory + requires the appropriate password or private key for decryption prior to + viewing any files, or any information about the files, in the archive. + + 7.1.10 Older ZIP compatible programs not familiar with the Central Directory + Encryption feature will no longer be able to recognize the Central + Directory and MAY assume the ZIP file is corrupt. Programs that + attempt streaming access using Local Headers will see invalid + information for each file. Central Directory Encryption need not be + used for every ZIP file. Its use is recommended for greater security. + ZIP files not using Central Directory Encryption SHOULD operate as + in the past. + + 7.1.11 This strong encryption feature specification is intended to provide for + scalable, cross-platform encryption needs ranging from simple password + encryption to authenticated public/private key encryption. + + 7.1.12 Encryption provides data confidentiality and privacy. It is + recommended that you combine X.509 digital signing with encryption + to add authentication and non-repudiation. + + +7.2 Single Password Symmetric Encryption Method +----------------------------------------------- + + 7.2.1 The Single Password Symmetric Encryption Method using strong + encryption algorithms operates similarly to the traditional + PKWARE encryption defined in this format. Additional data + structures are added to support the processing needs of the + strong algorithms. + + The Strong Encryption data structures are: + + 7.2.2 General Purpose Bits - Bits 0 and 6 of the General Purpose bit + flag in both local and central header records. Both bits set + indicates strong encryption. Bit 13, when set indicates the Central + Directory is encrypted and that selected fields in the Local Header + are masked to hide their actual value. + + + 7.2.3 Extra Field 0x0017 in central header only. + + Fields to consider in this record are: + + 7.2.3.1 Format - the data format identifier for this record. The only + value allowed at this time is the integer value 2. + + 7.2.3.2 AlgId - integer identifier of the encryption algorithm from the + following range + + 0x6601 - DES + 0x6602 - RC2 (version needed to extract < 5.2) + 0x6603 - 3DES 168 + 0x6609 - 3DES 112 + 0x660E - AES 128 + 0x660F - AES 192 + 0x6610 - AES 256 + 0x6702 - RC2 (version needed to extract >= 5.2) + 0x6720 - Blowfish + 0x6721 - Twofish + 0x6801 - RC4 + 0xFFFF - Unknown algorithm + + 7.2.3.3 Bitlen - Explicit bit length of key + + 32 - 448 bits + + 7.2.3.4 Flags - Processing flags needed for decryption + + 0x0001 - Password is required to decrypt + 0x0002 - Certificates only + 0x0003 - Password or certificate required to decrypt + + Values > 0x0003 reserved for certificate processing + + + 7.2.4 Decryption header record preceding compressed file data. + + -Decryption Header: + + Value Size Description + ----- ---- ----------- + IVSize 2 bytes Size of initialization vector (IV) + IVData IVSize Initialization vector for this file + Size 4 bytes Size of remaining decryption header data + Format 2 bytes Format definition for this record + AlgID 2 bytes Encryption algorithm identifier + Bitlen 2 bytes Bit length of encryption key + Flags 2 bytes Processing flags + ErdSize 2 bytes Size of Encrypted Random Data + ErdData ErdSize Encrypted Random Data + Reserved1 4 bytes Reserved certificate processing data + Reserved2 (var) Reserved for certificate processing data + VSize 2 bytes Size of password validation data + VData VSize-4 Password validation data + VCRC32 4 bytes Standard ZIP CRC32 of password validation data + + 7.2.4.1 IVData - The size of the IV SHOULD match the algorithm block size. + The IVData can be completely random data. If the size of + the randomly generated data does not match the block size + it SHOULD be complemented with zero's or truncated as + necessary. If IVSize is 0,then IV = CRC32 + Uncompressed + File Size (as a 64 bit little-endian, unsigned integer value). + + 7.2.4.2 Format - the data format identifier for this record. The only + value allowed at this time is the integer value 3. + + 7.2.4.3 AlgId - integer identifier of the encryption algorithm from the + following range + + 0x6601 - DES + 0x6602 - RC2 (version needed to extract < 5.2) + 0x6603 - 3DES 168 + 0x6609 - 3DES 112 + 0x660E - AES 128 + 0x660F - AES 192 + 0x6610 - AES 256 + 0x6702 - RC2 (version needed to extract >= 5.2) + 0x6720 - Blowfish + 0x6721 - Twofish + 0x6801 - RC4 + 0xFFFF - Unknown algorithm + + 7.2.4.4 Bitlen - Explicit bit length of key + + 32 - 448 bits + + 7.2.4.5 Flags - Processing flags needed for decryption + + 0x0001 - Password is required to decrypt + 0x0002 - Certificates only + 0x0003 - Password or certificate required to decrypt + + Values > 0x0003 reserved for certificate processing + + 7.2.4.6 ErdData - Encrypted random data is used to store random data that + is used to generate a file session key for encrypting + each file. SHA1 is used to calculate hash data used to + derive keys. File session keys are derived from a master + session key generated from the user-supplied password. + If the Flags field in the decryption header contains + the value 0x4000, then the ErdData field MUST be + decrypted using 3DES. If the value 0x4000 is not set, + then the ErdData field MUST be decrypted using AlgId. + + + 7.2.4.7 Reserved1 - Reserved for certificate processing, if value is + zero, then Reserved2 data is absent. See the explanation + under the Certificate Processing Method for details on + this data structure. + + 7.2.4.8 Reserved2 - If present, the size of the Reserved2 data structure + is located by skipping the first 4 bytes of this field + and using the next 2 bytes as the remaining size. See + the explanation under the Certificate Processing Method + for details on this data structure. + + 7.2.4.9 VSize - This size value will always include the 4 bytes of the + VCRC32 data and will be greater than 4 bytes. + + 7.2.4.10 VData - Random data for password validation. This data is VSize + in length and VSize MUST be a multiple of the encryption + block size. VCRC32 is a checksum value of VData. + VData and VCRC32 are stored encrypted and start the + stream of encrypted data for a file. + + + 7.2.5 Useful Tips + + 7.2.5.1 Strong Encryption is always applied to a file after compression. The + block oriented algorithms all operate in Cypher Block Chaining (CBC) + mode. The block size used for AES encryption is 16. All other block + algorithms use a block size of 8. Two IDs are defined for RC2 to + account for a discrepancy found in the implementation of the RC2 + algorithm in the cryptographic library on Windows XP SP1 and all + earlier versions of Windows. It is recommended that zero length files + not be encrypted, however programs SHOULD be prepared to extract them + if they are found within a ZIP file. + + 7.2.5.2 A pseudo-code representation of the encryption process is as follows: + + Password = GetUserPassword() + MasterSessionKey = DeriveKey(SHA1(Password)) + RD = CryptographicStrengthRandomData() + For Each File + IV = CryptographicStrengthRandomData() + VData = CryptographicStrengthRandomData() + VCRC32 = CRC32(VData) + FileSessionKey = DeriveKey(SHA1(IV + RD) + ErdData = Encrypt(RD,MasterSessionKey,IV) + Encrypt(VData + VCRC32 + FileData, FileSessionKey,IV) + Done + + 7.2.5.3 The function names and parameter requirements will depend on + the choice of the cryptographic toolkit selected. Almost any + toolkit supporting the reference implementations for each + algorithm can be used. The RSA BSAFE(r), OpenSSL, and Microsoft + CryptoAPI libraries are all known to work well. + + + 7.3 Single Password - Central Directory Encryption + -------------------------------------------------- + + 7.3.1 Central Directory Encryption is achieved within the .ZIP format by + encrypting the Central Directory structure. This encapsulates the metadata + most often used for processing .ZIP files. Additional metadata is stored for + redundancy in the Local Header for each file. The process of concealing + metadata by encrypting the Central Directory does not protect the data within + the Local Header. To avoid information leakage from the exposed metadata + in the Local Header, the fields containing information about a file are masked. + + 7.3.2 Local Header + + Masking replaces the true content of the fields for a file in the Local + Header with false information. When masked, the Local Header is not + suitable for streaming access and the options for data recovery of damaged + archives is reduced. Extra Data fields that MAY contain confidential + data SHOULD NOT be stored within the Local Header. The value set into + the Version needed to extract field SHOULD be the correct value needed to + extract the file without regard to Central Directory Encryption. The fields + within the Local Header targeted for masking when the Central Directory is + encrypted are: + + Field Name Mask Value + ------------------ --------------------------- + compression method 0 + last mod file time 0 + last mod file date 0 + crc-32 0 + compressed size 0 + uncompressed size 0 + file name (variable size) Base 16 value from the + range 1 - 0xFFFFFFFFFFFFFFFF + represented as a string whose + size will be set into the + file name length field + + The Base 16 value assigned as a masked file name is simply a sequentially + incremented value for each file starting with 1 for the first file. + Modifications to a ZIP file MAY cause different values to be stored for + each file. For compatibility, the file name field in the Local Header + SHOULD NOT be left blank. As of Version 6.2 of this specification, + the Compression Method and Compressed Size fields are not yet masked. + Fields having a value of 0xFFFF or 0xFFFFFFFF for the ZIP64 format + SHOULD NOT be masked. + + 7.3.3 Encrypting the Central Directory + + Encryption of the Central Directory does not include encryption of the + Central Directory Signature data, the Zip64 End of Central Directory + record, the Zip64 End of Central Directory Locator, or the End + of Central Directory record. The ZIP file comment data is never + encrypted. + + Before encrypting the Central Directory, it MAY optionally be compressed. + Compression is not required, but for storage efficiency it is assumed + this structure will be compressed before encrypting. Similarly, this + specification supports compressing the Central Directory without + requiring that it also be encrypted. Early implementations of this + feature will assume the encryption method applied to files matches the + encryption applied to the Central Directory. + + Encryption of the Central Directory is done in a manner similar to + that of file encryption. The encrypted data is preceded by a + decryption header. The decryption header is known as the Archive + Decryption Header. The fields of this record are identical to + the decryption header preceding each encrypted file. The location + of the Archive Decryption Header is determined by the value in the + Start of the Central Directory field in the Zip64 End of Central + Directory record. When the Central Directory is encrypted, the + Zip64 End of Central Directory record will always be present. + + The layout of the Zip64 End of Central Directory record for all + versions starting with 6.2 of this specification will follow the + Version 2 format. The Version 2 format is as follows: + + The leading fixed size fields within the Version 1 format for this + record remain unchanged. The record signature for both Version 1 + and Version 2 will be 0x06064b50. Immediately following the last + byte of the field known as the Offset of Start of Central + Directory With Respect to the Starting Disk Number will begin the + new fields defining Version 2 of this record. + + 7.3.4 New fields for Version 2 + + Note: all fields stored in Intel low-byte/high-byte order. + + Value Size Description + ----- ---- ----------- + Compression Method 2 bytes Method used to compress the + Central Directory + Compressed Size 8 bytes Size of the compressed data + Original Size 8 bytes Original uncompressed size + AlgId 2 bytes Encryption algorithm ID + BitLen 2 bytes Encryption key length + Flags 2 bytes Encryption flags + HashID 2 bytes Hash algorithm identifier + Hash Length 2 bytes Length of hash data + Hash Data (variable) Hash data + + The Compression Method accepts the same range of values as the + corresponding field in the Central Header. + + The Compressed Size and Original Size values will not include the + data of the Central Directory Signature which is compressed or + encrypted. + + The AlgId, BitLen, and Flags fields accept the same range of values + the corresponding fields within the 0x0017 record. + + Hash ID identifies the algorithm used to hash the Central Directory + data. This data does not have to be hashed, in which case the + values for both the HashID and Hash Length will be 0. Possible + values for HashID are: + + Value Algorithm + ------ --------- + 0x0000 none + 0x0001 CRC32 + 0x8003 MD5 + 0x8004 SHA1 + 0x8007 RIPEMD160 + 0x800C SHA256 + 0x800D SHA384 + 0x800E SHA512 + + 7.3.5 When the Central Directory data is signed, the same hash algorithm + used to hash the Central Directory for signing SHOULD be used. + This is recommended for processing efficiency, however, it is + permissible for any of the above algorithms to be used independent + of the signing process. + + The Hash Data will contain the hash data for the Central Directory. + The length of this data will vary depending on the algorithm used. + + The Version Needed to Extract SHOULD be set to 62. + + The value for the Total Number of Entries on the Current Disk will + be 0. These records will no longer support random access when + encrypting the Central Directory. + + 7.3.6 When the Central Directory is compressed and/or encrypted, the + End of Central Directory record will store the value 0xFFFFFFFF + as the value for the Total Number of Entries in the Central + Directory. The value stored in the Total Number of Entries in + the Central Directory on this Disk field will be 0. The actual + values will be stored in the equivalent fields of the Zip64 + End of Central Directory record. + + 7.3.7 Decrypting and decompressing the Central Directory is accomplished + in the same manner as decrypting and decompressing a file. + + 7.4 Certificate Processing Method + --------------------------------- + + The Certificate Processing Method for ZIP file encryption + defines the following additional data fields: + + 7.4.1 Certificate Flag Values + + Additional processing flags that can be present in the Flags field of both + the 0x0017 field of the central directory Extra Field and the Decryption + header record preceding compressed file data are: + + 0x0007 - reserved for future use + 0x000F - reserved for future use + 0x0100 - Indicates non-OAEP key wrapping was used. If this + this field is set, the version needed to extract MUST + be at least 61. This means OAEP key wrapping is not + used when generating a Master Session Key using + ErdData. + 0x4000 - ErdData MUST be decrypted using 3DES-168, otherwise use the + same algorithm used for encrypting the file contents. + 0x8000 - reserved for future use + + + 7.4.2 CertData - Extra Field 0x0017 record certificate data structure + + The data structure used to store certificate data within the section + of the Extra Field defined by the CertData field of the 0x0017 + record are as shown: + + Value Size Description + ----- ---- ----------- + RCount 4 bytes Number of recipients. + HashAlg 2 bytes Hash algorithm identifier + HSize 2 bytes Hash size + SRList (var) Simple list of recipients hashed public keys + + + RCount This defines the number intended recipients whose + public keys were used for encryption. This identifies + the number of elements in the SRList. + + HashAlg This defines the hash algorithm used to calculate + the public key hash of each public key used + for encryption. This field currently supports + only the following value for SHA-1 + + 0x8004 - SHA1 + + HSize This defines the size of a hashed public key. + + SRList This is a variable length list of the hashed + public keys for each intended recipient. Each + element in this list is HSize. The total size of + SRList is determined using RCount * HSize. + + + 7.4.3 Reserved1 - Certificate Decryption Header Reserved1 Data + + Value Size Description + ----- ---- ----------- + RCount 4 bytes Number of recipients. + + RCount This defines the number intended recipients whose + public keys were used for encryption. This defines + the number of elements in the REList field defined below. + + + 7.4.4 Reserved2 - Certificate Decryption Header Reserved2 Data Structures + + + Value Size Description + ----- ---- ----------- + HashAlg 2 bytes Hash algorithm identifier + HSize 2 bytes Hash size + REList (var) List of recipient data elements + + + HashAlg This defines the hash algorithm used to calculate + the public key hash of each public key used + for encryption. This field currently supports + only the following value for SHA-1 + + 0x8004 - SHA1 + + HSize This defines the size of a hashed public key + defined in REHData. + + REList This is a variable length of list of recipient data. + Each element in this list consists of a Recipient + Element data structure as follows: + + + Recipient Element (REList) Data Structure: + + Value Size Description + ----- ---- ----------- + RESize 2 bytes Size of REHData + REKData + REHData HSize Hash of recipients public key + REKData (var) Simple key blob + + + RESize This defines the size of an individual REList + element. This value is the combined size of the + REHData field + REKData field. REHData is defined by + HSize. REKData is variable and can be calculated + for each REList element using RESize and HSize. + + REHData Hashed public key for this recipient. + + REKData Simple Key Blob. The format of this data structure + is identical to that defined in the Microsoft + CryptoAPI and generated using the CryptExportKey() + function. The version of the Simple Key Blob + supported at this time is 0x02 as defined by + Microsoft. + +7.5 Certificate Processing - Central Directory Encryption +--------------------------------------------------------- + + 7.5.1 Central Directory Encryption using Digital Certificates will + operate in a manner similar to that of Single Password Central + Directory Encryption. This record will only be present when there + is data to place into it. Currently, data is placed into this + record when digital certificates are used for either encrypting + or signing the files within a ZIP file. When only password + encryption is used with no certificate encryption or digital + signing, this record is not currently needed. When present, this + record will appear before the start of the actual Central Directory + data structure and will be located immediately after the Archive + Decryption Header if the Central Directory is encrypted. + + 7.5.2 The Archive Extra Data record will be used to store the following + information. Additional data MAY be added in future versions. + + Extra Data Fields: + + 0x0014 - PKCS#7 Store for X.509 Certificates + 0x0016 - X.509 Certificate ID and Signature for central directory + 0x0019 - PKCS#7 Encryption Recipient Certificate List + + The 0x0014 and 0x0016 Extra Data records that otherwise would be + located in the first record of the Central Directory for digital + certificate processing. When encrypting or compressing the Central + Directory, the 0x0014 and 0x0016 records MUST be located in the + Archive Extra Data record and they SHOULD NOT remain in the first + Central Directory record. The Archive Extra Data record will also + be used to store the 0x0019 data. + + 7.5.3 When present, the size of the Archive Extra Data record will be + included in the size of the Central Directory. The data of the + Archive Extra Data record will also be compressed and encrypted + along with the Central Directory data structure. + +7.6 Certificate Processing Differences +-------------------------------------- + + 7.6.1 The Certificate Processing Method of encryption differs from the + Single Password Symmetric Encryption Method as follows. Instead + of using a user-defined password to generate a master session key, + cryptographically random data is used. The key material is then + wrapped using standard key-wrapping techniques. This key material + is wrapped using the public key of each recipient that will need + to decrypt the file using their corresponding private key. + + 7.6.2 This specification currently assumes digital certificates will follow + the X.509 V3 format for 1024 bit and higher RSA format digital + certificates. Implementation of this Certificate Processing Method + requires supporting logic for key access and management. This logic + is outside the scope of this specification. + +7.7 OAEP Processing with Certificate-based Encryption +----------------------------------------------------- + + 7.7.1 OAEP stands for Optimal Asymmetric Encryption Padding. It is a + strengthening technique used for small encoded items such as decryption + keys. This is commonly applied in cryptographic key-wrapping techniques + and is supported by PKCS #1. Versions 5.0 and 6.0 of this specification + were designed to support OAEP key-wrapping for certificate-based + decryption keys for additional security. + + 7.7.2 Support for private keys stored on Smartcards or Tokens introduced + a conflict with this OAEP logic. Most card and token products do + not support the additional strengthening applied to OAEP key-wrapped + data. In order to resolve this conflict, versions 6.1 and above of this + specification will no longer support OAEP when encrypting using + digital certificates. + + 7.7.3 Versions of PKZIP available during initial development of the + certificate processing method set a value of 61 into the + version needed to extract field for a file. This indicates that + non-OAEP key wrapping is used. This affects certificate encryption + only, and password encryption functions SHOULD NOT be affected by + this value. This means values of 61 MAY be found on files encrypted + with certificates only, or on files encrypted with both password + encryption and certificate encryption. Files encrypted with both + methods can safely be decrypted using the password methods documented. + +7.8 Additional Encryption/Decryption Data Records +----------------------------------------------------- + + 7.8.1 Additional information MAY be stored within a ZIP file in support + of the strong password and certificate encryption methods defined above. + These include, but are not limited to the following record types. + + 0x0021 Policy Decryption Key Record + 0x0022 Smartcrypt Key Provider Record + 0x0023 Smartcrypt Policy Key Data Record + +8.0 Splitting and Spanning ZIP files +------------------------------------- + + 8.1 Spanned ZIP files + + 8.1.1 Spanning is the process of segmenting a ZIP file across + multiple removable media. This support has typically only + been provided for DOS formatted floppy diskettes. + + 8.2 Split ZIP files + + 8.2.1 File splitting is a newer derivation of spanning. + Splitting follows the same segmentation process as + spanning, however, it does not require writing each + segment to a unique removable medium and instead supports + placing all pieces onto local or non-removable locations + such as file systems, local drives, folders, etc. + + 8.3 File Naming Differences + + 8.3.1 A key difference between spanned and split ZIP files is + that all pieces of a spanned ZIP file have the same name. + Since each piece is written to a separate volume, no name + collisions occur and each segment can reuse the original + .ZIP file name given to the archive. + + 8.3.2 Sequence ordering for DOS spanned archives uses the DOS + volume label to determine segment numbers. Volume labels + for each segment are written using the form PKBACK#xxx, + where xxx is the segment number written as a decimal + value from 001 - nnn. + + 8.3.3 Split ZIP files are typically written to the same location + and are subject to name collisions if the spanned name + format is used since each segment will reside on the same + drive. To avoid name collisions, split archives are named + as follows. + + Segment 1 = filename.z01 + Segment n-1 = filename.z(n-1) + Segment n = filename.zip + + 8.3.4 The .ZIP extension is used on the last segment to support + quickly reading the central directory. The segment number + n SHOULD be a decimal value. + + 8.4 Spanned Self-extracting ZIP Files + + 8.4.1 Spanned ZIP files MAY be PKSFX Self-extracting ZIP files. + PKSFX files MAY also be split, however, in this case + the first segment MUST be named filename.exe. The first + segment of a split PKSFX archive MUST be large enough to + include the entire executable program. + + 8.5 Capacities and Markers + + 8.5.1 Capacities for split archives are as follows: + + Maximum number of segments = 4,294,967,295 - 1 + Maximum .ZIP segment size = 4,294,967,295 bytes + Minimum segment size = 64K + Maximum PKSFX segment size = 2,147,483,647 bytes + + 8.5.2 Segment sizes MAY be different however by convention, all + segment sizes SHOULD be the same with the exception of the + last, which MAY be smaller. Local and central directory + header records MUST NOT be split across a segment boundary. + When writing a header record, if the number of bytes remaining + within a segment is less than the size of the header record, + end the current segment and write the header at the start + of the next segment. The central directory MAY span segment + boundaries, but no single record in the central directory + SHOULD be split across segments. + + 8.5.3 Spanned/Split archives created using PKZIP for Windows + (V2.50 or greater), PKZIP Command Line (V2.50 or greater), + or PKZIP Explorer will include a special spanning + signature as the first 4 bytes of the first segment of + the archive. This signature (0x08074b50) will be + followed immediately by the local header signature for + the first file in the archive. + + 8.5.4 A special spanning marker MAY also appear in spanned/split + archives if the spanning or splitting process starts but + only requires one segment. In this case the 0x08074b50 + signature will be replaced with the temporary spanning + marker signature of 0x30304b50. Split archives can + only be uncompressed by other versions of PKZIP that + know how to create a split archive. + + 8.5.5 The signature value 0x08074b50 is also used by some + ZIP implementations as a marker for the Data Descriptor + record. Conflict in this alternate assignment can be + avoided by ensuring the position of the signature + within the ZIP file to determine the use for which it + is intended. + +9.0 Change Process +------------------ + + 9.1 In order for the .ZIP file format to remain a viable technology, this + specification SHOULD be considered as open for periodic review and + revision. Although this format was originally designed with a + certain level of extensibility, not all changes in technology + (present or future) were or will be necessarily considered in its + design. + + 9.2 If your application requires new definitions to the + extensible sections in this format, or if you would like to + submit new data structures or new capabilities, please forward + your request to zipformat@pkware.com. All submissions will be + reviewed by the ZIP File Specification Committee for possible + inclusion into future versions of this specification. + + 9.3 Periodic revisions to this specification will be published as + DRAFT or as FINAL status to ensure interoperability. We encourage + comments and feedback that MAY help improve clarity or content. + + +10.0 Incorporating PKWARE Proprietary Technology into Your Product +------------------------------------------------------------------ + + 10.1 The Use or Implementation in a product of APPNOTE technological + components pertaining to either strong encryption or patching requires + a separate, executed license agreement from PKWARE. Please contact + PKWARE at zipformat@pkware.com or +1-414-289-9788 with regard to + acquiring such a license. + + 10.2 Additional information regarding PKWARE proprietary technology is + available at http://www.pkware.com/appnote. + +11.0 Acknowledgements +--------------------- + + In addition to the above mentioned contributors to PKZIP and PKUNZIP, + PKWARE would like to extend special thanks to Robert Mahoney for + suggesting the extension .ZIP for this software. + +12.0 References +--------------- + + Fiala, Edward R., and Greene, Daniel H., "Data compression with + finite windows", Communications of the ACM, Volume 32, Number 4, + April 1989, pages 490-505. + + Held, Gilbert, "Data Compression, Techniques and Applications, + Hardware and Software Considerations", John Wiley & Sons, 1987. + + Huffman, D.A., "A method for the construction of minimum-redundancy + codes", Proceedings of the IRE, Volume 40, Number 9, September 1952, + pages 1098-1101. + + Nelson, Mark, "LZW Data Compression", Dr. Dobbs Journal, Volume 14, + Number 10, October 1989, pages 29-37. + + Nelson, Mark, "The Data Compression Book", M&T Books, 1991. + + Storer, James A., "Data Compression, Methods and Theory", + Computer Science Press, 1988 + + Welch, Terry, "A Technique for High-Performance Data Compression", + IEEE Computer, Volume 17, Number 6, June 1984, pages 8-19. + + Ziv, J. and Lempel, A., "A universal algorithm for sequential data + compression", Communications of the ACM, Volume 30, Number 6, + June 1987, pages 520-540. + + Ziv, J. and Lempel, A., "Compression of individual sequences via + variable-rate coding", IEEE Transactions on Information Theory, + Volume 24, Number 5, September 1978, pages 530-536. + + +APPENDIX A - AS/400 Extra Field (0x0065) Attribute Definitions +-------------------------------------------------------------- + +A.1 Field Definition Structure: + + a. field length including length 2 bytes Big Endian + b. field code 2 bytes + c. data x bytes + +A.2 Field Code Description + + 4001 Source type i.e. CLP etc + 4002 The text description of the library + 4003 The text description of the file + 4004 The text description of the member + 4005 x'F0' or 0 is PF-DTA, x'F1' or 1 is PF_SRC + 4007 Database Type Code 1 byte + 4008 Database file and fields definition + 4009 GZIP file type 2 bytes + 400B IFS code page 2 bytes + 400C IFS Time of last file status change 4 bytes + 400D IFS Access Time 4 bytes + 400E IFS Modification time 4 bytes + 005C Length of the records in the file 2 bytes + 0068 GZIP two words 8 bytes + +APPENDIX B - z/OS Extra Field (0x0065) Attribute Definitions +------------------------------------------------------------ + +B.1 Field Definition Structure: + + a. field length including length 2 bytes Big Endian + b. field code 2 bytes + c. data x bytes + +B.2 Field Code Description + + 0001 File Type 2 bytes + 0002 NonVSAM Record Format 1 byte + 0003 Reserved + 0004 NonVSAM Block Size 2 bytes Big Endian + 0005 Primary Space Allocation 3 bytes Big Endian + 0006 Secondary Space Allocation 3 bytes Big Endian + 0007 Space Allocation Type1 byte flag + 0008 Modification Date Retired with PKZIP 5.0 + + 0009 Expiration Date Retired with PKZIP 5.0 + + 000A PDS Directory Block Allocation 3 bytes Big Endian binary value + 000B NonVSAM Volume List variable + 000C UNIT Reference Retired with PKZIP 5.0 + + 000D DF/SMS Management Class 8 bytes EBCDIC Text Value + 000E DF/SMS Storage Class 8 bytes EBCDIC Text Value + 000F DF/SMS Data Class 8 bytes EBCDIC Text Value + 0010 PDS/PDSE Member Info. 30 bytes + 0011 VSAM sub-filetype 2 bytes + 0012 VSAM LRECL 13 bytes EBCDIC "(num_avg num_max)" + 0013 VSAM Cluster Name Retired with PKZIP 5.0 + + 0014 VSAM KSDS Key Information 13 bytes EBCDIC "(num_length num_position)" + 0015 VSAM Average LRECL 5 bytes EBCDIC num_value padded with blanks + 0016 VSAM Maximum LRECL 5 bytes EBCDIC num_value padded with blanks + 0017 VSAM KSDS Key Length 5 bytes EBCDIC num_value padded with blanks + 0018 VSAM KSDS Key Position 5 bytes EBCDIC num_value padded with blanks + 0019 VSAM Data Name 1-44 bytes EBCDIC text string + 001A VSAM KSDS Index Name 1-44 bytes EBCDIC text string + 001B VSAM Catalog Name 1-44 bytes EBCDIC text string + 001C VSAM Data Space Type 9 bytes EBCDIC text string + 001D VSAM Data Space Primary 9 bytes EBCDIC num_value left-justified + 001E VSAM Data Space Secondary 9 bytes EBCDIC num_value left-justified + 001F VSAM Data Volume List variable EBCDIC text list of 6-character Volume IDs + 0020 VSAM Data Buffer Space 8 bytes EBCDIC num_value left-justified + 0021 VSAM Data CISIZE 5 bytes EBCDIC num_value left-justified + 0022 VSAM Erase Flag 1 byte flag + 0023 VSAM Free CI % 3 bytes EBCDIC num_value left-justified + 0024 VSAM Free CA % 3 bytes EBCDIC num_value left-justified + 0025 VSAM Index Volume List variable EBCDIC text list of 6-character Volume IDs + 0026 VSAM Ordered Flag 1 byte flag + 0027 VSAM REUSE Flag 1 byte flag + 0028 VSAM SPANNED Flag 1 byte flag + 0029 VSAM Recovery Flag 1 byte flag + 002A VSAM WRITECHK Flag 1 byte flag + 002B VSAM Cluster/Data SHROPTS 3 bytes EBCDIC "n,y" + 002C VSAM Index SHROPTS 3 bytes EBCDIC "n,y" + 002D VSAM Index Space Type 9 bytes EBCDIC text string + 002E VSAM Index Space Primary 9 bytes EBCDIC num_value left-justified + 002F VSAM Index Space Secondary 9 bytes EBCDIC num_value left-justified + 0030 VSAM Index CISIZE 5 bytes EBCDIC num_value left-justified + 0031 VSAM Index IMBED 1 byte flag + 0032 VSAM Index Ordered Flag 1 byte flag + 0033 VSAM REPLICATE Flag 1 byte flag + 0034 VSAM Index REUSE Flag 1 byte flag + 0035 VSAM Index WRITECHK Flag 1 byte flag Retired with PKZIP 5.0 + + 0036 VSAM Owner 8 bytes EBCDIC text string + 0037 VSAM Index Owner 8 bytes EBCDIC text string + 0038 Reserved + 0039 Reserved + 003A Reserved + 003B Reserved + 003C Reserved + 003D Reserved + 003E Reserved + 003F Reserved + 0040 Reserved + 0041 Reserved + 0042 Reserved + 0043 Reserved + 0044 Reserved + 0045 Reserved + 0046 Reserved + 0047 Reserved + 0048 Reserved + 0049 Reserved + 004A Reserved + 004B Reserved + 004C Reserved + 004D Reserved + 004E Reserved + 004F Reserved + 0050 Reserved + 0051 Reserved + 0052 Reserved + 0053 Reserved + 0054 Reserved + 0055 Reserved + 0056 Reserved + 0057 Reserved + 0058 PDS/PDSE Member TTR Info. 6 bytes Big Endian + 0059 PDS 1st LMOD Text TTR 3 bytes Big Endian + 005A PDS LMOD EP Rec # 4 bytes Big Endian + 005B Reserved + 005C Max Length of records 2 bytes Big Endian + 005D PDSE Flag 1 byte flag + 005E Reserved + 005F Reserved + 0060 Reserved + 0061 Reserved + 0062 Reserved + 0063 Reserved + 0064 Reserved + 0065 Last Date Referenced 4 bytes Packed Hex "yyyymmdd" + 0066 Date Created 4 bytes Packed Hex "yyyymmdd" + 0068 GZIP two words 8 bytes + 0071 Extended NOTE Location 12 bytes Big Endian + 0072 Archive device UNIT 6 bytes EBCDIC + 0073 Archive 1st Volume 6 bytes EBCDIC + 0074 Archive 1st VOL File Seq# 2 bytes Binary + 0075 Native I/O Flags 2 bytes + 0081 Unix File Type 1 byte enumerated + 0082 Unix File Format 1 byte enumerated + 0083 Unix File Character Set Tag Info 4 bytes + 0090 ZIP Environmental Processing Info 4 bytes + 0091 EAV EATTR Flags 1 byte + 0092 DSNTYPE Flags 1 byte + 0093 Total Space Allocation (Cyls) 4 bytes Big Endian + 009D NONVSAM DSORG 2 bytes + 009E Program Virtual Object Info 3 bytes + 009F Encapsulated file Info 9 bytes + 400C Unix File Creation Time 4 bytes + 400D Unix File Access Time 4 bytes + 400E Unix File Modification time 4 bytes + 4101 IBMCMPSC Compression Info variable + 4102 IBMCMPSC Compression Size 8 bytes Big Endian + +APPENDIX C - Zip64 Extensible Data Sector Mappings +--------------------------------------------------- + + -Z390 Extra Field: + + The following is the general layout of the attributes for the + ZIP 64 "extra" block for extended tape operations. + + Note: some fields stored in Big Endian format. All text is + in EBCDIC format unless otherwise specified. + + Value Size Description + ----- ---- ----------- + (Z390) 0x0065 2 bytes Tag for this "extra" block type + Size 4 bytes Size for the following data block + Tag 4 bytes EBCDIC "Z390" + Length71 2 bytes Big Endian + Subcode71 2 bytes Enote type code + FMEPos 1 byte + Length72 2 bytes Big Endian + Subcode72 2 bytes Unit type code + Unit 1 byte Unit + Length73 2 bytes Big Endian + Subcode73 2 bytes Volume1 type code + FirstVol 1 byte Volume + Length74 2 bytes Big Endian + Subcode74 2 bytes FirstVol file sequence + FileSeq 2 bytes Sequence + +APPENDIX D - Language Encoding (EFS) +------------------------------------ + +D.1 The ZIP format has historically supported only the original IBM PC character +encoding set, commonly referred to as IBM Code Page 437. This limits storing +file name characters to only those within the original MS-DOS range of values +and does not properly support file names in other character encodings, or +languages. To address this limitation, this specification will support the +following change. + +D.2 If general purpose bit 11 is unset, the file name and comment SHOULD conform +to the original ZIP character encoding. If general purpose bit 11 is set, the +filename and comment MUST support The Unicode Standard, Version 4.1.0 or +greater using the character encoding form defined by the UTF-8 storage +specification. The Unicode Standard is published by the The Unicode +Consortium (www.unicode.org). UTF-8 encoded data stored within ZIP files +is expected to not include a byte order mark (BOM). + +D.3 Applications MAY choose to supplement this file name storage through the use +of the 0x0008 Extra Field. Storage for this optional field is currently +undefined, however it will be used to allow storing extended information +on source or target encoding that MAY further assist applications with file +name, or file content encoding tasks. Please contact PKWARE with any +requirements on how this field SHOULD be used. + +D.4 The 0x0008 Extra Field storage MAY be used with either setting for general +purpose bit 11. Examples of the intended usage for this field is to store +whether "modified-UTF-8" (JAVA) is used, or UTF-8-MAC. Similarly, other +commonly used character encoding (code page) designations can be indicated +through this field. Formalized values for use of the 0x0008 record remain +undefined at this time. The definition for the layout of the 0x0008 field +will be published when available. Use of the 0x0008 Extra Field provides +for storing data within a ZIP file in an encoding other than IBM Code +Page 437 or UTF-8. + +D.5 General purpose bit 11 will not imply any encoding of file content or +password. Values defining character encoding for file content or +password MUST be stored within the 0x0008 Extended Language Encoding +Extra Field. + +D.6 Ed Gordon of the Info-ZIP group has defined a pair of "extra field" records +that can be used to store UTF-8 file name and file comment fields. These +records can be used for cases when the general purpose bit 11 method +for storing UTF-8 data in the standard file name and comment fields is +not desirable. A common case for this alternate method is if backward +compatibility with older programs is required. + +D.7 Definitions for the record structure of these fields are included above +in the section on 3rd party mappings for "extra field" records. These +records are identified by Header ID's 0x6375 (Info-ZIP Unicode Comment +Extra Field) and 0x7075 (Info-ZIP Unicode Path Extra Field). + +D.8 The choice of which storage method to use when writing a ZIP file is left +to the implementation. Developers SHOULD expect that a ZIP file MAY +contain either method and SHOULD provide support for reading data in +either format. Use of general purpose bit 11 reduces storage requirements +for file name data by not requiring additional "extra field" data for +each file, but can result in older ZIP programs not being able to extract +files. Use of the 0x6375 and 0x7075 records will result in a ZIP file +that SHOULD always be readable by older ZIP programs, but requires more +storage per file to write file name and/or file comment fields. + +APPENDIX E - AE-x encryption marker +----------------------------------- + +E.1 AE-x defines an alternate password-based encryption method used +in ZIP files that is based on a file encryption utility developed by +Dr. Brian Gladman. Information on Dr. Gladman's method is available at + + http://www.gladman.me.uk/cryptography_technology/fileencrypt/ + +E.2 AE-x uses AES with CTR (counter mode) and HMAC-SHA1. It defines +encryption using key sizes of 128 bits or 256 bits. It does not +restrict support for decrypting 192 bits. + +E.3 This method uses the standard ZIP encryption bit (bit 0) +of the general purpose bit flag (section 4.4.4) to indicate a +file is encrypted. + +E.4 The compression method field (section 4.4.5) is set to 99 +to indicate a file has been encrypted using this method. + +E.5 The actual compression method is stored in an extra field +structure identified by a Header ID of 0x9901. Information on this +record structure can be found at http://www.winzip.com/aes_info.htm. + +E.6 Two versions are defined for the 0x9901 structure. + + E.6.1 Version 1 stores the file CRC value in the CRC-32 field + (section 4.4.7). + + E.6.2 Version 2 stores a value of 0 in the CRC-32 field. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..e2c3a91 --- /dev/null +++ b/LICENSE @@ -0,0 +1,18 @@ +MIT License + +Copyright (c) 2026 JunHo + +Permission is hereby granted, free of charge, to any person obtaining a copy of this software and +associated documentation files (the "Software"), to deal in the Software without restriction, including +without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the +following conditions: + +The above copyright notice and this permission notice shall be included in all copies or substantial +portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT +LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO +EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE +USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/MyZip.sln b/MyZip.sln new file mode 100644 index 0000000..cd4f817 --- /dev/null +++ b/MyZip.sln @@ -0,0 +1,31 @@ + +Microsoft Visual Studio Solution File, Format Version 12.00 +# Visual Studio Version 17 +VisualStudioVersion = 17.14.37516.0 d17.14 +MinimumVisualStudioVersion = 10.0.40219.1 +Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "MyZip", "MyZip.vcxproj", "{31C9539B-F71A-4637-862A-D14876B0E120}" +EndProject +Global + GlobalSection(SolutionConfigurationPlatforms) = preSolution + Debug|x64 = Debug|x64 + Debug|x86 = Debug|x86 + Release|x64 = Release|x64 + Release|x86 = Release|x86 + EndGlobalSection + GlobalSection(ProjectConfigurationPlatforms) = postSolution + {31C9539B-F71A-4637-862A-D14876B0E120}.Debug|x64.ActiveCfg = Debug|x64 + {31C9539B-F71A-4637-862A-D14876B0E120}.Debug|x64.Build.0 = Debug|x64 + {31C9539B-F71A-4637-862A-D14876B0E120}.Debug|x86.ActiveCfg = Debug|Win32 + {31C9539B-F71A-4637-862A-D14876B0E120}.Debug|x86.Build.0 = Debug|Win32 + {31C9539B-F71A-4637-862A-D14876B0E120}.Release|x64.ActiveCfg = Release|x64 + {31C9539B-F71A-4637-862A-D14876B0E120}.Release|x64.Build.0 = Release|x64 + {31C9539B-F71A-4637-862A-D14876B0E120}.Release|x86.ActiveCfg = Release|Win32 + {31C9539B-F71A-4637-862A-D14876B0E120}.Release|x86.Build.0 = Release|Win32 + EndGlobalSection + GlobalSection(SolutionProperties) = preSolution + HideSolutionNode = FALSE + EndGlobalSection + GlobalSection(ExtensibilityGlobals) = postSolution + SolutionGuid = {65D8C98E-0364-4B64-A488-3E30F4C0F3A2} + EndGlobalSection +EndGlobal diff --git a/MyZip.vcxproj b/MyZip.vcxproj new file mode 100644 index 0000000..3a71371 --- /dev/null +++ b/MyZip.vcxproj @@ -0,0 +1,131 @@ + + + + + Debug + Win32 + + + Release + Win32 + + + Debug + x64 + + + Release + x64 + + + + 17.0 + Win32Proj + {31c9539b-f71a-4637-862a-d14876b0e120} + MyZip + 10.0 + + + + Application + true + v143 + Unicode + + + Application + false + v143 + true + Unicode + + + Application + true + v143 + Unicode + + + Application + false + v143 + true + Unicode + + + + + + + + + + + + + + + + + + + + + + Level3 + true + WIN32;_DEBUG;_CONSOLE;%(PreprocessorDefinitions) + true + + + Console + true + + + + + Level3 + true + true + true + WIN32;NDEBUG;_CONSOLE;%(PreprocessorDefinitions) + true + + + Console + true + + + + + Level3 + true + _DEBUG;_CONSOLE;%(PreprocessorDefinitions) + true + + + Console + true + + + + + Level3 + true + true + true + NDEBUG;_CONSOLE;%(PreprocessorDefinitions) + true + + + Console + true + + + + + + + + + \ No newline at end of file diff --git a/MyZip.vcxproj.filters b/MyZip.vcxproj.filters new file mode 100644 index 0000000..2c1b3dc --- /dev/null +++ b/MyZip.vcxproj.filters @@ -0,0 +1,22 @@ + + + + + {4FC737F1-C7A5-4376-A066-2A32D752A2FF} + cpp;c;cc;cxx;c++;cppm;ixx;def;odl;idl;hpj;bat;asm;asmx + + + {93995380-89BD-4b04-88EB-625FBE52EBFB} + h;hh;hpp;hxx;h++;hm;inl;inc;ipp;xsd + + + {67DA6AB6-F800-4c08-8B7A-83BB121AAD01} + rc;ico;cur;bmp;dlg;rc2;rct;bin;rgs;gif;jpg;jpeg;jpe;resx;tiff;tif;png;wav;mfcribbon-ms + + + + + 소스 파일 + + + \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..7cb6276 --- /dev/null +++ b/README.md @@ -0,0 +1,3 @@ +# MyZip + +Make Zip without AI Assist \ No newline at end of file diff --git a/main.cpp b/main.cpp new file mode 100644 index 0000000..c493a39 --- /dev/null +++ b/main.cpp @@ -0,0 +1,6 @@ +#include + +int main() +{ + std::printf("Hello Zip! \r\n"); +} \ No newline at end of file