v2
This commit is contained in:
+374
@@ -0,0 +1,374 @@
|
|||||||
|
"""
|
||||||
|
PST Email Extractor
|
||||||
|
====================
|
||||||
|
Extracts email metadata from Outlook PST files into CSV format.
|
||||||
|
Supports batch processing of multiple PST files in a directory.
|
||||||
|
|
||||||
|
Dependencies:
|
||||||
|
pip install tqdm libpff-python
|
||||||
|
(see README.md for platform-specific libpff/pypff installation)
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python Extract.py [directory]
|
||||||
|
If no directory is given, the script's own directory is used.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import pypff
|
||||||
|
import csv
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
import glob
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
from functools import lru_cache
|
||||||
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
|
||||||
|
# ─────────────────────────── helpers ────────────────────────────
|
||||||
|
|
||||||
|
def _safe_str(value, max_len: int = 0) -> str:
|
||||||
|
"""Convert a value to a clean string, optionally truncating it."""
|
||||||
|
try:
|
||||||
|
result = "" if value is None else str(value)
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
return result[:max_len] if max_len else result
|
||||||
|
|
||||||
|
|
||||||
|
def get_folder_path(folder) -> str:
|
||||||
|
"""
|
||||||
|
Walk up the parent chain to reconstruct the full folder path.
|
||||||
|
Capped at 50 levels to prevent runaway recursion on corrupt files.
|
||||||
|
"""
|
||||||
|
parts: list[str] = []
|
||||||
|
current = folder
|
||||||
|
for _ in range(50):
|
||||||
|
if current is None:
|
||||||
|
break
|
||||||
|
try:
|
||||||
|
name = getattr(current, "name", None)
|
||||||
|
parts.insert(0, str(name) if name else "Unknown_Folder")
|
||||||
|
current = getattr(current, "parent", None)
|
||||||
|
except Exception:
|
||||||
|
break
|
||||||
|
clean = [p for p in parts if p]
|
||||||
|
return "\\".join(clean) if clean else "Root"
|
||||||
|
|
||||||
|
|
||||||
|
def _format_dt(dt_obj) -> str:
|
||||||
|
"""Return an ISO-8601 string for a datetime-like object, UTC-aware."""
|
||||||
|
if dt_obj is None:
|
||||||
|
return ""
|
||||||
|
try:
|
||||||
|
# pypff delivery/submit times are naive UTC – make them explicit
|
||||||
|
if dt_obj.tzinfo is None:
|
||||||
|
dt_obj = dt_obj.replace(tzinfo=timezone.utc)
|
||||||
|
return dt_obj.strftime("%Y-%m-%d %H:%M:%S UTC")
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
# ──────────────────────── email extraction ──────────────────────
|
||||||
|
|
||||||
|
def extract_email_info(message, folder_path: str) -> dict:
|
||||||
|
"""
|
||||||
|
Pull all useful metadata from a pypff message object.
|
||||||
|
Every field is individually guarded so one bad attribute never
|
||||||
|
kills an entire message.
|
||||||
|
"""
|
||||||
|
# Subject ──────────────────────────────────────────────────────
|
||||||
|
subject = _safe_str(getattr(message, "subject", ""), max_len=500)
|
||||||
|
|
||||||
|
# Sender ───────────────────────────────────────────────────────
|
||||||
|
try:
|
||||||
|
sender_name = _safe_str(getattr(message, "sender_name", ""))
|
||||||
|
sender_email = _safe_str(getattr(message, "sender_email_address", ""))
|
||||||
|
if sender_name and sender_email:
|
||||||
|
sender = f"{sender_name} <{sender_email}>"
|
||||||
|
else:
|
||||||
|
sender = sender_name or sender_email or "Unknown Sender"
|
||||||
|
except Exception:
|
||||||
|
sender = "Unknown Sender"
|
||||||
|
sender = sender[:200]
|
||||||
|
|
||||||
|
# Recipients ───────────────────────────────────────────────────
|
||||||
|
to_field = ""
|
||||||
|
try:
|
||||||
|
count = getattr(message, "recipients_count", 0) or 0
|
||||||
|
recipients: list[str] = []
|
||||||
|
for i in range(min(count, 100)):
|
||||||
|
try:
|
||||||
|
r = message.recipients[i]
|
||||||
|
addr = _safe_str(getattr(r, "email_address", ""))
|
||||||
|
name = _safe_str(getattr(r, "name", ""))
|
||||||
|
recipients.append(addr or name)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
to_field = "; ".join(filter(None, recipients))[:1000]
|
||||||
|
except Exception:
|
||||||
|
to_field = ""
|
||||||
|
|
||||||
|
# Date ─────────────────────────────────────────────────────────
|
||||||
|
date_str = (
|
||||||
|
_format_dt(getattr(message, "delivery_time", None))
|
||||||
|
or _format_dt(getattr(message, "client_submit_time", None))
|
||||||
|
)
|
||||||
|
|
||||||
|
# Size (bytes) ─────────────────────────────────────────────────
|
||||||
|
size = 0
|
||||||
|
for attr in ("plain_text_body", "html_body", "transport_headers", "subject"):
|
||||||
|
try:
|
||||||
|
content = getattr(message, attr, None)
|
||||||
|
if content:
|
||||||
|
# pypff often returns bytes; len() works on both str and bytes
|
||||||
|
size += len(content)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Attachments count ────────────────────────────────────────────
|
||||||
|
attachments = 0
|
||||||
|
try:
|
||||||
|
attachments = getattr(message, "number_of_attachments", 0) or 0
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Message class (e.g. IPM.Note, IPM.Appointment …) ────────────
|
||||||
|
msg_class = _safe_str(getattr(message, "message_class", ""), max_len=100)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"Folder": str(folder_path)[:200],
|
||||||
|
"Subject": subject,
|
||||||
|
"From": sender,
|
||||||
|
"To": to_field,
|
||||||
|
"Date": date_str,
|
||||||
|
"Size": str(size),
|
||||||
|
"Attachments": str(attachments),
|
||||||
|
"MessageClass": msg_class,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ────────────────────────── counting ────────────────────────────
|
||||||
|
|
||||||
|
def count_total_messages(folder) -> int:
|
||||||
|
"""Recursively count every message in the folder tree (for the progress bar)."""
|
||||||
|
total = 0
|
||||||
|
try:
|
||||||
|
total += getattr(folder, "number_of_sub_messages", 0) or 0
|
||||||
|
for i in range(getattr(folder, "number_of_sub_folders", 0) or 0):
|
||||||
|
try:
|
||||||
|
sub = folder.get_sub_folder(i)
|
||||||
|
if sub:
|
||||||
|
total += count_total_messages(sub)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return total
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── folder walk ──────────────────────────
|
||||||
|
|
||||||
|
def process_folder(folder, csv_writer, pbar, counters: dict):
|
||||||
|
"""
|
||||||
|
Recursively walk all folders and write one CSV row per message.
|
||||||
|
Uses an iterative approach for subfolders to avoid Python recursion limits
|
||||||
|
on deeply nested PST files.
|
||||||
|
"""
|
||||||
|
folder_stack = [folder]
|
||||||
|
|
||||||
|
while folder_stack:
|
||||||
|
current = folder_stack.pop()
|
||||||
|
try:
|
||||||
|
folder_path = get_folder_path(current)
|
||||||
|
except Exception:
|
||||||
|
folder_path = "Unknown"
|
||||||
|
|
||||||
|
# ── messages in this folder ──
|
||||||
|
n_msgs = getattr(current, "number_of_sub_messages", 0) or 0
|
||||||
|
for i in range(n_msgs):
|
||||||
|
try:
|
||||||
|
msg = current.get_sub_message(i)
|
||||||
|
if msg is None:
|
||||||
|
continue
|
||||||
|
info = extract_email_info(msg, folder_path)
|
||||||
|
csv_writer.writerow([
|
||||||
|
info["Folder"],
|
||||||
|
info["Subject"],
|
||||||
|
info["From"],
|
||||||
|
info["To"],
|
||||||
|
info["Date"],
|
||||||
|
info["Size"],
|
||||||
|
info["Attachments"],
|
||||||
|
info["MessageClass"],
|
||||||
|
])
|
||||||
|
counters["ok"] += 1
|
||||||
|
except Exception:
|
||||||
|
counters["err"] += 1
|
||||||
|
finally:
|
||||||
|
pbar.update(1)
|
||||||
|
|
||||||
|
# ── queue subfolders ──
|
||||||
|
n_subs = getattr(current, "number_of_sub_folders", 0) or 0
|
||||||
|
for i in range(n_subs):
|
||||||
|
try:
|
||||||
|
sub = current.get_sub_folder(i)
|
||||||
|
if sub:
|
||||||
|
folder_stack.append(sub)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── PST export ───────────────────────────
|
||||||
|
|
||||||
|
FIELDNAMES = ["Folder", "Subject", "From", "To", "Date", "Size", "Attachments", "MessageClass"]
|
||||||
|
|
||||||
|
|
||||||
|
def export_pst_to_csv(pst_file_path: str, output_file_path: str) -> bool:
|
||||||
|
"""
|
||||||
|
Open a single PST file, walk its entire folder tree, and write metadata
|
||||||
|
for every message to a UTF-8 CSV file.
|
||||||
|
|
||||||
|
Returns True on success, False on failure.
|
||||||
|
"""
|
||||||
|
if not os.path.exists(pst_file_path):
|
||||||
|
print(f" ❌ PST file not found: {pst_file_path}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
# ── open PST ──────────────────────────────────────────────────
|
||||||
|
try:
|
||||||
|
pst = pypff.file()
|
||||||
|
pst.open(pst_file_path)
|
||||||
|
except Exception as exc:
|
||||||
|
print(f" ❌ Cannot open PST: {exc}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
try:
|
||||||
|
root = pst.get_root_folder()
|
||||||
|
if root is None:
|
||||||
|
print(" ❌ Cannot access root folder")
|
||||||
|
return False
|
||||||
|
|
||||||
|
base = os.path.basename(pst_file_path)
|
||||||
|
print(f" 📊 Counting messages in {base} …")
|
||||||
|
total = count_total_messages(root)
|
||||||
|
print(f" ✓ {total:,} messages found")
|
||||||
|
|
||||||
|
counters = {"ok": 0, "err": 0}
|
||||||
|
|
||||||
|
with open(output_file_path, "w", newline="", encoding="utf-8") as fh:
|
||||||
|
writer = csv.writer(fh)
|
||||||
|
writer.writerow(FIELDNAMES)
|
||||||
|
|
||||||
|
bar_fmt = "{desc}: {percentage:3.0f}%|{bar}| {n_fmt}/{total_fmt} [{elapsed}<{remaining}]"
|
||||||
|
with tqdm(total=total, desc=f" Processing {base}", unit="email",
|
||||||
|
bar_format=bar_fmt) as pbar:
|
||||||
|
process_folder(root, writer, pbar, counters)
|
||||||
|
|
||||||
|
print(f" ✓ Written {counters['ok']:,} rows | Errors skipped: {counters['err']:,}")
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as exc:
|
||||||
|
print(f" ❌ Critical error: {exc}")
|
||||||
|
return False
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
pst.close()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
# ────────────────────── batch processing ────────────────────────
|
||||||
|
|
||||||
|
def find_pst_files(directory: str) -> list[str]:
|
||||||
|
"""Return sorted list of .pst files in *directory* (non-recursive)."""
|
||||||
|
return sorted(glob.glob(os.path.join(directory, "*.pst")))
|
||||||
|
|
||||||
|
|
||||||
|
def process_all_pst_files(directory: str) -> None:
|
||||||
|
"""Discover and process every PST file found in *directory*."""
|
||||||
|
sep = "=" * 80
|
||||||
|
|
||||||
|
print(sep)
|
||||||
|
print("🔍 SEARCHING FOR PST FILES")
|
||||||
|
print(sep)
|
||||||
|
|
||||||
|
pst_files = find_pst_files(directory)
|
||||||
|
if not pst_files:
|
||||||
|
print(f"❌ No PST files found in: {directory}")
|
||||||
|
return
|
||||||
|
|
||||||
|
total_files = len(pst_files)
|
||||||
|
print(f"✓ Found {total_files} PST file(s)")
|
||||||
|
print()
|
||||||
|
print("📁 PST Files:")
|
||||||
|
for idx, f in enumerate(pst_files, 1):
|
||||||
|
size_mb = os.path.getsize(f) / (1024 ** 2)
|
||||||
|
print(f" {idx:>3}. {os.path.basename(f)} ({size_mb:.1f} MB)")
|
||||||
|
print()
|
||||||
|
|
||||||
|
print(sep)
|
||||||
|
print("🚀 PROCESSING")
|
||||||
|
print(sep)
|
||||||
|
|
||||||
|
ok_count = 0
|
||||||
|
fail_count = 0
|
||||||
|
|
||||||
|
for idx, pst_file in enumerate(pst_files, 1):
|
||||||
|
remaining = total_files - idx
|
||||||
|
print()
|
||||||
|
print("─" * 80)
|
||||||
|
print(f"📊 {idx}/{total_files} | ✓ Done: {ok_count} | ❌ Failed: {fail_count} | ⏳ Remaining: {remaining}")
|
||||||
|
print(f"🔄 {os.path.basename(pst_file)}")
|
||||||
|
|
||||||
|
base_name = os.path.splitext(os.path.basename(pst_file))[0]
|
||||||
|
output_file = os.path.join(directory, f"{base_name}_email_list.csv")
|
||||||
|
|
||||||
|
if export_pst_to_csv(pst_file, output_file):
|
||||||
|
ok_count += 1
|
||||||
|
rows = sum(1 for _ in open(output_file, encoding="utf-8")) - 1 # minus header
|
||||||
|
print(f" ✅ Saved → {os.path.basename(output_file)} ({rows:,} rows)")
|
||||||
|
else:
|
||||||
|
fail_count += 1
|
||||||
|
print(" ❌ Failed")
|
||||||
|
|
||||||
|
# ── summary ───────────────────────────────────────────────────
|
||||||
|
print()
|
||||||
|
print(sep)
|
||||||
|
print("📈 FINAL SUMMARY")
|
||||||
|
print(sep)
|
||||||
|
print(f" PST files found: {total_files}")
|
||||||
|
print(f" ✓ Processed: {ok_count}")
|
||||||
|
print(f" ❌ Failed: {fail_count}")
|
||||||
|
if total_files:
|
||||||
|
print(f" Success rate: {ok_count / total_files * 100:.1f}%")
|
||||||
|
print(sep)
|
||||||
|
|
||||||
|
if ok_count == total_files:
|
||||||
|
print("🎉 All files processed successfully!")
|
||||||
|
elif ok_count:
|
||||||
|
print("⚠️ Some files failed – check errors above.")
|
||||||
|
else:
|
||||||
|
print("❌ No files were processed. Check errors above.")
|
||||||
|
print(sep)
|
||||||
|
|
||||||
|
|
||||||
|
# ─────────────────────────── entry point ────────────────────────
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
if len(sys.argv) > 1:
|
||||||
|
directory = sys.argv[1]
|
||||||
|
else:
|
||||||
|
directory = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
print(f"ℹ️ No directory specified – using script location: {directory}")
|
||||||
|
print(f"ℹ️ Tip: python {os.path.basename(__file__)} <directory_path>")
|
||||||
|
print()
|
||||||
|
|
||||||
|
if not os.path.isdir(directory):
|
||||||
|
print(f"❌ Not a valid directory: {directory}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
print(f"📂 Working directory: {directory}")
|
||||||
|
print("ℹ️ Corrupt or unsupported messages are skipped automatically.")
|
||||||
|
print()
|
||||||
|
|
||||||
|
process_all_pst_files(directory)
|
||||||
@@ -0,0 +1,280 @@
|
|||||||
|
# PST Email Extractor
|
||||||
|
|
||||||
|
A command-line tool that scans a directory for Outlook PST files and extracts the metadata of every email — subject, sender, recipients, date, size, attachments count, and message class — into individual CSV files.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Table of Contents
|
||||||
|
|
||||||
|
1. [Features](#features)
|
||||||
|
2. [Requirements](#requirements)
|
||||||
|
3. [Installation](#installation)
|
||||||
|
4. [Usage](#usage)
|
||||||
|
5. [Output Format](#output-format)
|
||||||
|
6. [How It Works](#how-it-works)
|
||||||
|
7. [Performance Notes](#performance-notes)
|
||||||
|
8. [Troubleshooting](#troubleshooting)
|
||||||
|
9. [Platform-specific pypff Installation](#platform-specific-pypff-installation)
|
||||||
|
10. [Project Structure](#project-structure)
|
||||||
|
11. [FAQ](#faq)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
- **Batch processing** – drop any number of `.pst` files in a folder and run once
|
||||||
|
- **Resilient extraction** – corrupt messages are skipped gracefully; processing continues
|
||||||
|
- **Rich metadata** – exports 8 fields per message including attachment count and message class
|
||||||
|
- **Progress bars** – real-time per-file progress via `tqdm`
|
||||||
|
- **UTF-8 output** – fully compatible with Excel, Google Sheets, and any modern CSV tool
|
||||||
|
- **Iterative folder traversal** – avoids Python recursion limits on deeply nested PST archives
|
||||||
|
- **Human-friendly summaries** – per-run statistics on success / failure counts
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
| Dependency | Version | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| Python | ≥ 3.8 | Runtime |
|
||||||
|
| `tqdm` | ≥ 4.66 | Progress bars |
|
||||||
|
| `pypff` / `libpff-python` | any | PST file reading |
|
||||||
|
|
||||||
|
> **Note:** `pypff` wraps the native C library **libpff**. Installation varies by platform — see [Platform-specific pypff Installation](#platform-specific-pypff-installation).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
### Step 1 – Clone or download the project
|
||||||
|
|
||||||
|
Place `Extract.py` and `install_dependencies.py` in the same directory.
|
||||||
|
|
||||||
|
### Step 2 – Run the dependency installer
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python install_dependencies.py
|
||||||
|
```
|
||||||
|
|
||||||
|
This script will:
|
||||||
|
|
||||||
|
1. Upgrade `pip`, `setuptools`, and `wheel`.
|
||||||
|
2. Install `tqdm` via pip.
|
||||||
|
3. Attempt to install `libpff-python` / `pypff` via pip.
|
||||||
|
4. Print platform-specific instructions if the automatic install fails.
|
||||||
|
|
||||||
|
### Step 3 – Verify
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -c "import pypff; import tqdm; print('All good!')"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
### Basic usage (process PST files in the script's own directory)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python Extract.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### Specify a directory
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python Extract.py /path/to/pst/folder
|
||||||
|
```
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
# Windows PowerShell
|
||||||
|
python Extract.py "C:\Users\You\Desktop\PST_Files"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Example output
|
||||||
|
|
||||||
|
```
|
||||||
|
📂 Working directory: /data/pst_files
|
||||||
|
|
||||||
|
================================================================================
|
||||||
|
🔍 SEARCHING FOR PST FILES
|
||||||
|
================================================================================
|
||||||
|
✓ Found 3 PST file(s)
|
||||||
|
|
||||||
|
📁 PST Files:
|
||||||
|
1. archive_2021.pst (1,204.3 MB)
|
||||||
|
2. archive_2022.pst (876.1 MB)
|
||||||
|
3. personal.pst (312.7 MB)
|
||||||
|
|
||||||
|
================================================================================
|
||||||
|
🚀 PROCESSING
|
||||||
|
================================================================================
|
||||||
|
|
||||||
|
────────────────────────────────────────────────────────────────────────────────
|
||||||
|
📊 1/3 | ✓ Done: 0 | ❌ Failed: 0 | ⏳ Remaining: 2
|
||||||
|
🔄 archive_2021.pst
|
||||||
|
📊 Counting messages in archive_2021.pst …
|
||||||
|
✓ 42,817 messages found
|
||||||
|
Processing archive_2021.pst: 100%|████████| 42817/42817 [02:14<00:00]
|
||||||
|
✓ Written 42,817 rows | Errors skipped: 3
|
||||||
|
✅ Saved → archive_2021_email_list.csv (42,817 rows)
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Output Format
|
||||||
|
|
||||||
|
For each input file `<name>.pst`, a corresponding `<name>_email_list.csv` is created in the **same directory**.
|
||||||
|
|
||||||
|
### CSV columns
|
||||||
|
|
||||||
|
| Column | Description | Max Length |
|
||||||
|
|---|---|---|
|
||||||
|
| `Folder` | Full folder path inside the PST (e.g. `Inbox\Projects\2022`) | 200 chars |
|
||||||
|
| `Subject` | Email subject line | 500 chars |
|
||||||
|
| `From` | Sender name and/or email address | 200 chars |
|
||||||
|
| `To` | Semicolon-separated recipient list | 1000 chars |
|
||||||
|
| `Date` | Delivery or submit time in `YYYY-MM-DD HH:MM:SS UTC` | — |
|
||||||
|
| `Size` | Estimated message size in bytes (sum of body and headers) | — |
|
||||||
|
| `Attachments` | Number of attachments | — |
|
||||||
|
| `MessageClass` | MAPI message class (e.g. `IPM.Note`, `IPM.Appointment`) | 100 chars |
|
||||||
|
|
||||||
|
### Example rows
|
||||||
|
|
||||||
|
```csv
|
||||||
|
Folder,Subject,From,To,Date,Size,Attachments,MessageClass
|
||||||
|
Inbox,Q3 Budget Review,Alice Smith <alice@example.com>,bob@example.com,2023-09-01 14:22:10 UTC,18432,2,IPM.Note
|
||||||
|
Sent Items,Re: Proposal,Bob Jones <bob@example.com>,alice@example.com,2023-09-02 09:05:44 UTC,4210,0,IPM.Note
|
||||||
|
Calendar,,Alice Smith <alice@example.com>,,2023-10-10 09:00:00 UTC,1024,0,IPM.Appointment
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
1. **Discovery** – `find_pst_files()` uses `glob` to locate all `.pst` files in the given directory.
|
||||||
|
2. **Counting** – Before extraction, `count_total_messages()` recursively counts all messages so `tqdm` can show an accurate progress bar.
|
||||||
|
3. **Traversal** – `process_folder()` uses an **iterative stack** (not Python recursion) to walk the full folder tree, guarding against deep nesting and corrupt nodes.
|
||||||
|
4. **Extraction** – `extract_email_info()` wraps every individual attribute access in a `try/except` so one corrupt property never aborts the whole message.
|
||||||
|
5. **Writing** – Results are streamed directly to a UTF-8 CSV file via `csv.writer`, keeping memory usage flat regardless of PST size.
|
||||||
|
6. **Error recovery** – Failed messages are counted and reported in the summary but do not interrupt processing.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Performance Notes
|
||||||
|
|
||||||
|
- **Memory** – The extractor streams rows directly to disk, so even a 50 GB PST file won't exhaust RAM.
|
||||||
|
- **Speed** – Processing speed depends on your disk I/O. SSDs can process ~5,000–15,000 messages/min; spinning disks are 3–5× slower.
|
||||||
|
- **Large recipient lists** – Recipient lists are capped at 100 entries and 1,000 characters to avoid giant rows.
|
||||||
|
- **Iterative traversal** – The folder walker uses an explicit stack instead of recursion, making it safe for PST archives with hundreds of nested folders.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
### `ModuleNotFoundError: No module named 'pypff'`
|
||||||
|
|
||||||
|
Run `python install_dependencies.py`. If the automatic install fails, see [Platform-specific pypff Installation](#platform-specific-pypff-installation).
|
||||||
|
|
||||||
|
### `Failed to open PST file`
|
||||||
|
|
||||||
|
- The file may be in use by Outlook. Close Outlook completely and try again.
|
||||||
|
- The file may be corrupted. Try running the Outlook Inbox Repair Tool (`scanpst.exe` on Windows).
|
||||||
|
|
||||||
|
### Progress bar reaches 100% but fewer rows are written than expected
|
||||||
|
|
||||||
|
This is normal. Some items in a PST folder hierarchy are not email messages (calendar events, contacts, tasks). They are counted in the total but may fail extraction silently. The `Errors skipped` count in the summary reflects this.
|
||||||
|
|
||||||
|
### CSV opens with garbled characters in Excel
|
||||||
|
|
||||||
|
The CSV is UTF-8. In Excel, use **Data → From Text/CSV** and choose **UTF-8** encoding, or open via `File → Open` and select the encoding in the import wizard.
|
||||||
|
|
||||||
|
### Very slow processing on large PST files
|
||||||
|
|
||||||
|
Ensure the PST file is on a local drive, not a network share or cloud-synced folder. Network latency multiplies enormously across hundreds of thousands of small reads.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Platform-specific pypff Installation
|
||||||
|
|
||||||
|
### Linux (Debian / Ubuntu)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo apt-get install python3-libpff
|
||||||
|
# or
|
||||||
|
pip install libpff-python
|
||||||
|
```
|
||||||
|
|
||||||
|
### Linux (Fedora / RHEL)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo dnf install libpff-devel python3-libpff
|
||||||
|
```
|
||||||
|
|
||||||
|
### macOS (Homebrew)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
brew install libpff
|
||||||
|
pip install libpff-python
|
||||||
|
```
|
||||||
|
|
||||||
|
### Windows
|
||||||
|
|
||||||
|
Automatic pip installation may fail because libpff requires compilation.
|
||||||
|
|
||||||
|
**Option 1 – Pre-built wheel** (if available for your Python version):
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
pip install libpff-python
|
||||||
|
```
|
||||||
|
|
||||||
|
**Option 2 – Build from source** – follow the official guide:
|
||||||
|
https://github.com/libyal/libpff/blob/main/documentation/Building.md
|
||||||
|
|
||||||
|
**Option 3 – WSL (recommended)** – Install Windows Subsystem for Linux, then follow the Ubuntu steps above. This is the easiest path for most Windows users.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Project Structure
|
||||||
|
|
||||||
|
```
|
||||||
|
.
|
||||||
|
├── Extract.py # Main extractor – run this
|
||||||
|
├── install_dependencies.py # One-time dependency installer
|
||||||
|
└── README.md # This file
|
||||||
|
```
|
||||||
|
|
||||||
|
Outputs are placed alongside the input PST files:
|
||||||
|
|
||||||
|
```
|
||||||
|
/your/pst/folder/
|
||||||
|
├── archive_2021.pst
|
||||||
|
├── archive_2021_email_list.csv ← generated
|
||||||
|
├── archive_2022.pst
|
||||||
|
└── archive_2022_email_list.csv ← generated
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## FAQ
|
||||||
|
|
||||||
|
**Does the tool modify the PST files?**
|
||||||
|
No. PST files are opened read-only. The tool never writes to them.
|
||||||
|
|
||||||
|
**Can I run it on a single PST file instead of a whole directory?**
|
||||||
|
Put the PST file in its own folder and pass that folder path. Alternatively, the function `export_pst_to_csv(pst_path, output_csv_path)` can be imported and called directly from your own script.
|
||||||
|
|
||||||
|
**What happens if I run it twice on the same directory?**
|
||||||
|
Existing CSV files are overwritten without warning. If you want to preserve previous results, move or rename them first.
|
||||||
|
|
||||||
|
**Does it export email bodies or attachments?**
|
||||||
|
No — only metadata is exported. This keeps the output small and avoids legal / privacy issues associated with exporting full message content.
|
||||||
|
|
||||||
|
**What message types are included?**
|
||||||
|
All MAPI message objects found in the PST: emails (`IPM.Note`), meeting requests (`IPM.Schedule.*`), appointments (`IPM.Appointment`), contacts (`IPM.Contact`), tasks (`IPM.Task`), etc. The `MessageClass` column lets you filter by type in your spreadsheet.
|
||||||
|
|
||||||
|
**Will it work with OST files?**
|
||||||
|
libpff has partial OST support. Rename the file to `.pst` and try — results may vary depending on the OST version and whether the file is currently synced.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
*For bug reports and feature requests, open an issue in your project repository.*
|
||||||
@@ -0,0 +1,176 @@
|
|||||||
|
"""
|
||||||
|
install_dependencies.py
|
||||||
|
========================
|
||||||
|
Installs all Python dependencies for the PST Email Extractor and
|
||||||
|
verifies that pypff (libpff) is importable.
|
||||||
|
|
||||||
|
Run once before first use:
|
||||||
|
python install_dependencies.py
|
||||||
|
|
||||||
|
Platform notes
|
||||||
|
--------------
|
||||||
|
* Linux / macOS – pypff is usually available via pip (libpff-python) or a
|
||||||
|
system package manager.
|
||||||
|
* Windows – pypff typically requires a pre-built wheel or compiling
|
||||||
|
libpff from source. See the README for details.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import subprocess
|
||||||
|
import platform
|
||||||
|
from importlib.util import find_spec
|
||||||
|
|
||||||
|
# ── package lists ────────────────────────────────────────────────
|
||||||
|
# Packages that install cleanly via pip on all platforms
|
||||||
|
PIP_PACKAGES: list[tuple[str, str]] = [
|
||||||
|
("tqdm", "tqdm>=4.66"), # Progress bars
|
||||||
|
]
|
||||||
|
|
||||||
|
# pypff can be installed with 'libpff-python' on Linux/macOS
|
||||||
|
PYPFF_PIP_NAME = "libpff-python"
|
||||||
|
|
||||||
|
|
||||||
|
# ── helpers ──────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def run(cmd: list[str]) -> subprocess.CompletedProcess:
|
||||||
|
"""Print and run a command, returning the result without raising."""
|
||||||
|
print(f"\n>> {' '.join(cmd)}")
|
||||||
|
return subprocess.run(cmd, check=False)
|
||||||
|
|
||||||
|
|
||||||
|
def is_importable(module: str) -> bool:
|
||||||
|
"""Return True if *module* can be found in the current environment."""
|
||||||
|
return find_spec(module) is not None
|
||||||
|
|
||||||
|
|
||||||
|
# ── steps ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def ensure_pip() -> None:
|
||||||
|
"""Bootstrap / upgrade pip, setuptools, and wheel."""
|
||||||
|
print("\n[1/3] Ensuring pip is up-to-date …")
|
||||||
|
run([sys.executable, "-m", "ensurepip", "--upgrade"])
|
||||||
|
run([sys.executable, "-m", "pip", "install", "--upgrade",
|
||||||
|
"pip", "setuptools", "wheel"])
|
||||||
|
|
||||||
|
|
||||||
|
def install_pip_packages() -> None:
|
||||||
|
"""Install the pip-only dependencies."""
|
||||||
|
print("\n[2/3] Installing pip packages …")
|
||||||
|
specs = [spec for _, spec in PIP_PACKAGES]
|
||||||
|
result = run([sys.executable, "-m", "pip", "install", "--upgrade"] + specs)
|
||||||
|
if result.returncode != 0:
|
||||||
|
print("\n❌ ERROR: Failed to install pip packages.")
|
||||||
|
sys.exit(result.returncode)
|
||||||
|
|
||||||
|
# Verify each import
|
||||||
|
for module, spec in PIP_PACKAGES:
|
||||||
|
ok = is_importable(module)
|
||||||
|
status = "✓" if ok else "✗"
|
||||||
|
print(f" {status} {spec} ({'ok' if ok else 'MISSING'})")
|
||||||
|
|
||||||
|
|
||||||
|
def install_pypff() -> bool:
|
||||||
|
"""
|
||||||
|
Try to install pypff via pip (works on Linux/macOS).
|
||||||
|
On Windows this usually fails; we give targeted guidance instead.
|
||||||
|
"""
|
||||||
|
print("\n[3/3] Installing pypff (libpff-python) …")
|
||||||
|
|
||||||
|
if is_importable("pypff"):
|
||||||
|
print(" ✓ pypff is already importable – skipping install.")
|
||||||
|
return True
|
||||||
|
|
||||||
|
result = run([sys.executable, "-m", "pip", "install", "--upgrade", PYPFF_PIP_NAME])
|
||||||
|
|
||||||
|
if result.returncode == 0 and is_importable("pypff"):
|
||||||
|
print(" ✓ pypff installed and importable.")
|
||||||
|
return True
|
||||||
|
|
||||||
|
# ── second attempt: some distros ship it as 'pypff' directly ──
|
||||||
|
result2 = run([sys.executable, "-m", "pip", "install", "--upgrade", "pypff"])
|
||||||
|
if result2.returncode == 0 and is_importable("pypff"):
|
||||||
|
print(" ✓ pypff installed via 'pypff' package name.")
|
||||||
|
return True
|
||||||
|
|
||||||
|
print(" ✗ Could not install pypff automatically.")
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# ── platform guidance ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
def print_pypff_guidance() -> None:
|
||||||
|
system = platform.system()
|
||||||
|
py_ver = f"{sys.version_info.major}.{sys.version_info.minor}"
|
||||||
|
|
||||||
|
print("\n" + "=" * 72)
|
||||||
|
print("MANUAL INSTALLATION REQUIRED: pypff / libpff")
|
||||||
|
print("=" * 72)
|
||||||
|
|
||||||
|
if system == "Linux":
|
||||||
|
print(
|
||||||
|
"On Debian / Ubuntu:\n"
|
||||||
|
" sudo apt-get install python3-libpff\n"
|
||||||
|
" # or\n"
|
||||||
|
" pip install libpff-python\n\n"
|
||||||
|
"On Fedora / RHEL:\n"
|
||||||
|
" sudo dnf install libpff-devel python3-libpff\n"
|
||||||
|
)
|
||||||
|
elif system == "Darwin":
|
||||||
|
print(
|
||||||
|
"On macOS (Homebrew):\n"
|
||||||
|
" brew install libpff\n"
|
||||||
|
" pip install libpff-python\n"
|
||||||
|
)
|
||||||
|
elif system == "Windows":
|
||||||
|
print(
|
||||||
|
f"On Windows (Python {py_ver}):\n"
|
||||||
|
" Option 1 – Pre-built wheel (if available for your Python version):\n"
|
||||||
|
" pip install libpff-python\n\n"
|
||||||
|
" Option 2 – Build from source:\n"
|
||||||
|
" https://github.com/libyal/libpff/blob/main/documentation/Building.md\n\n"
|
||||||
|
" Option 3 – Use WSL (Windows Subsystem for Linux) and follow the Linux steps.\n"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
print(
|
||||||
|
" Please visit https://github.com/libyal/libpff for build instructions\n"
|
||||||
|
f" appropriate for your platform ({system}).\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
print("=" * 72)
|
||||||
|
|
||||||
|
|
||||||
|
# ── main ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
print("PST Email Extractor – Dependency Installer")
|
||||||
|
print("=" * 72)
|
||||||
|
print(f" Python: {sys.version.replace(chr(10), ' ')}")
|
||||||
|
print(f" Executable: {sys.executable}")
|
||||||
|
print(f" Platform: {platform.platform()}")
|
||||||
|
print(f" CWD: {os.getcwd()}")
|
||||||
|
|
||||||
|
ensure_pip()
|
||||||
|
install_pip_packages()
|
||||||
|
has_pypff = install_pypff()
|
||||||
|
|
||||||
|
# ── summary ──────────────────────────────────────────────────
|
||||||
|
print("\n" + "=" * 72)
|
||||||
|
print("Installation summary")
|
||||||
|
print("=" * 72)
|
||||||
|
for module, spec in PIP_PACKAGES:
|
||||||
|
ok = is_importable(module)
|
||||||
|
print(f" {'✓' if ok else '✗'} {spec}")
|
||||||
|
print(f" {'✓' if has_pypff else '✗'} pypff (libpff)")
|
||||||
|
print("=" * 72)
|
||||||
|
|
||||||
|
if not has_pypff:
|
||||||
|
print_pypff_guidance()
|
||||||
|
sys.exit(2)
|
||||||
|
|
||||||
|
print("\n✅ All dependencies are installed.")
|
||||||
|
print(" You can now run: python Extract.py [directory]")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user