From 3f6fddc35ca692a8dc15b55bb97218233eafc4cc Mon Sep 17 00:00:00 2001 From: zhouyr9 Date: Mon, 5 Jan 2026 14:23:50 +0800 Subject: [PATCH] =?UTF-8?q?=E7=A7=BB=E9=99=A4OneNote=20tool?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/OneNote_tool/__init__.py | 28 --- src/OneNote_tool/onenote.py | 326 ----------------------------------- 2 files changed, 354 deletions(-) delete mode 100644 src/OneNote_tool/__init__.py delete mode 100644 src/OneNote_tool/onenote.py diff --git a/src/OneNote_tool/__init__.py b/src/OneNote_tool/__init__.py deleted file mode 100644 index e7b1c3b..0000000 --- a/src/OneNote_tool/__init__.py +++ /dev/null @@ -1,28 +0,0 @@ -"""OneNote_tool package exports utilities to load OneNote documents. - -Currently exposes load_OneNote_document. -""" -import importlib - -try: - # dynamic import to get reference if available - _mod = importlib.import_module(".onenote", __package__) - _load_OneNote_document = getattr(_mod, "load_OneNote_document", None) -except Exception: - _load_OneNote_document = None - - -def load_OneNote_document(*args, **kwargs): - """Load OneNote document (wrapper). - - Uses an eagerly-imported function when available (helps some IDEs/statics); otherwise - imports the implementation lazily to avoid import-time side effects. - """ - if _load_OneNote_document is not None: - return _load_OneNote_document(*args, **kwargs) - mod = importlib.import_module(".onenote", __package__) - func = getattr(mod, "load_OneNote_document") - return func(*args, **kwargs) - - -__all__ = ["load_OneNote_document"] diff --git a/src/OneNote_tool/onenote.py b/src/OneNote_tool/onenote.py deleted file mode 100644 index 6a8b068..0000000 --- a/src/OneNote_tool/onenote.py +++ /dev/null @@ -1,326 +0,0 @@ -"""OneNote loader utilities. - -Provides load_OneNote_document which supports: -- Loading from local exported OneNote HTML directory or single .html/.mht file. -- Loading via Microsoft Graph API when provided with "graph" source and an access token. - -This implementation keeps dependencies optional: only `requests` is required for Graph fetches. -""" -from typing import Optional, Dict, Any, List -import pathlib -import mimetypes -import logging -import re -import email -from email import policy - -logger = logging.getLogger(__name__) - -# Minimal contract: -# - input: path_or_source (str), options (dict) -# - output: Dict with keys: "type" ("local"|"graph"), "pages": list of {"title","content","id","lastModified"} - - -def _read_local_html_file(path: str) -> str: - with open(path, "rb") as f: - data = f.read() - # try to decode utf-8, fallback to latin1 - for enc in ("utf-8", "utf-8-sig", "latin-1"): - try: - return data.decode(enc) - except Exception: - continue - return data.decode("latin-1", errors="ignore") - - -def _decode_part_payload(part) -> str: - # part is an email.message.Message - raw = part.get_payload(decode=True) - if raw is None: - return "" - ch = part.get_content_charset() - if ch: - try: - return raw.decode(ch, errors="replace") - except Exception: - pass - for enc in ("utf-8", "utf-8-sig", "latin-1"): - try: - return raw.decode(enc) - except Exception: - continue - return raw.decode("latin-1", errors="replace") - - -def _extract_title_from_html(html: str) -> str: - m = re.search(r"(.*?)", html, flags=re.I | re.S) - if m: - return m.group(1).strip() - # try to find first h1/h2 as fallback - m2 = re.search(r"]*>(.*?)", html, flags=re.I | re.S) - if m2: - return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", "", m2.group(1))).strip() - return "(untitled)" - - -def _split_html_into_pages(html: str) -> List[str]: - """Heuristic split of a single HTML that may contain multiple OneNote pages. - - Tries multiple strategies in order and returns the first that yields more than one fragment: - 1. class-based:
- 2. id-based:
- 3. header-based: split by repeated

(or

) elements which often mark page titles - - Returns list of HTML fragments (each a standalone-ish HTML string) or single-element list if heuristics fail. - """ - # 1) class-based (existing) - pattern_class = re.compile(r"]+class=[\"'][^\"']*(?:page|onenote-page|onenotepage|oneNotePage)[^\"']*[\"'][^>]*>", flags=re.I) - matches = list(pattern_class.finditer(html)) - if len(matches) > 1: - parts = [] - head_match = re.search(r"(.*?)", html, flags=re.I | re.S) - head_html = head_match.group(0) if head_match else "" - for i, m in enumerate(matches): - start = m.start() - end = matches[i + 1].start() if i + 1 < len(matches) else len(html) - body_fragment = html[start:end] - frag = "" + head_html + "" + body_fragment + "" - parts.append(frag) - return parts - - # 2) id-based: look for divs whose id begins with 'Page' or 'page' or 'id="P' common OneNote exports - pattern_id = re.compile(r"]+id=[\"']?(?:Page|page|P)_[^\s\"'>]*[\"']?[^>]*>", flags=re.I) - matches_id = list(pattern_id.finditer(html)) - if len(matches_id) > 1: - parts = [] - head_match = re.search(r"(.*?)", html, flags=re.I | re.S) - head_html = head_match.group(0) if head_match else "" - for i, m in enumerate(matches_id): - start = m.start() - end = matches_id[i + 1].start() if i + 1 < len(matches_id) else len(html) - body_fragment = html[start:end] - frag = "" + head_html + "" + body_fragment + "" - parts.append(frag) - return parts - - # 3) header-based: split by repeated

(or

) which often denote page titles in OneNote exports - h1_matches = list(re.finditer(r"]*>", html, flags=re.I)) - if len(h1_matches) > 1: - parts = [] - head_match = re.search(r"(.*?)", html, flags=re.I | re.S) - head_html = head_match.group(0) if head_match else "" - for i, m in enumerate(h1_matches): - start = m.start() - end = h1_matches[i + 1].start() if i + 1 < len(h1_matches) else len(html) - body_fragment = html[start:end] - frag = "" + head_html + "" + body_fragment + "" - parts.append(frag) - return parts - - # fallback: no multi-page split detected - return [html] - - -def _parse_mht_bytes(data: bytes) -> List[Dict[str, Any]]: - """Parse MHT/MHTML bytes and return a list of pages (dicts with id/title/content). - - Strategy: - - Use email.parser.BytesParser to parse the multipart MHT. - - For each part with content-type text/html, treat as one page. - - If there's only a single html part but it seems to contain multiple pages, try heuristic splitting. - """ - msg = email.parser.BytesParser(policy=policy.default).parsebytes(data) - html_parts: List[str] = [] - ids: List[str] = [] - # walk parts - if msg.is_multipart(): - for part in msg.walk(): - ctype = part.get_content_type() - if ctype == "text/html": - try: - html = _decode_part_payload(part) - except Exception: - html = _decode_part_payload(part) - html_parts.append(html) - # try to get a useful id from Content-Location or Content-ID - loc = part.get("Content-Location") or part.get("Content-ID") or "" - ids.append(loc) - else: - # not multipart: try parse as text/html - ctype = msg.get_content_type() - if ctype == "text/html": - html_parts.append(_decode_part_payload(msg)) - ids.append("") - pages: List[Dict[str, Any]] = [] - if len(html_parts) == 0: - return pages - if len(html_parts) == 1: - # maybe this single html contains multiple OneNote pages; try splitting heuristically - subpages = _split_html_into_pages(html_parts[0]) - if len(subpages) > 1: - for idx, sp in enumerate(subpages): - title = _extract_title_from_html(sp) or f"page_{idx}" - pages.append({"id": ids[0] or f"mht_part_{idx}", "title": title, "content": sp}) - return pages - # otherwise single page - title = _extract_title_from_html(html_parts[0]) - pages.append({"id": ids[0] or "mht_root", "title": title, "content": html_parts[0]}) - return pages - # multiple html parts -> each is a page - for idx, html in enumerate(html_parts): - title = _extract_title_from_html(html) - pages.append({"id": ids[idx] if idx < len(ids) else f"mht_part_{idx}", "title": title, "content": html}) - return pages - - -def load_OneNote_document(path_or_source: str, options: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: - """Load OneNote content. - - If `path_or_source` is a local filesystem path to a directory or file, load local HTML/MHT exports. - If `path_or_source` == "graph", options must contain `access_token` and either `notebook_id` or `site_id` + `notebook_id`. - - Returns a dict with parsed pages. - """ - options = options or {} - - p = pathlib.Path(path_or_source) - if path_or_source == "graph": - # fetch via Microsoft Graph - token = options.get("access_token") - if not token: - raise ValueError("Graph source requires 'access_token' in options") - notebook_id = options.get("notebook_id") - # simple graph pagination for pages - import requests - - headers = {"Authorization": f"Bearer {token}"} - pages = [] - if not notebook_id: - # try current user's notebooks - url = "https://graph.microsoft.com/v1.0/me/onenote/notebooks" - else: - url = f"https://graph.microsoft.com/v1.0/me/onenote/notebooks/{notebook_id}/sections" - resp = requests.get(url, headers=headers) - resp.raise_for_status() - data = resp.json() - # This is a thin implementation: collect pages by iterating sections then pages - sections = data.get("value", []) - for sec in sections: - sec_id = sec.get("id") - sec_title = sec.get("displayName") - # list pages in section - pages_url = f"https://graph.microsoft.com/v1.0/me/onenote/sections/{sec_id}/pages" - r2 = requests.get(pages_url, headers=headers) - if r2.status_code != 200: - continue - for pjson in r2.json().get("value", []): - content_url = pjson.get("contentUrl") - page_id = pjson.get("id") - title = pjson.get("title") - last_modified = pjson.get("lastModifiedDateTime") - # fetch content (HTML) - try: - c_resp = requests.get(content_url, headers=headers) - c_resp.raise_for_status() - content = c_resp.text - except Exception as e: - logger.exception("Failed to fetch OneNote page content: %s", e) - content = "" - pages.append({"id": page_id, "title": title, "content": content, "lastModified": last_modified, "section": sec_title}) - return {"type": "graph", "pages": pages} - - # Local path handling - if not p.exists(): - raise FileNotFoundError(f"Path not found: {path_or_source}") - - pages = [] - if p.is_file(): - mtype, _ = mimetypes.guess_type(p.as_posix()) - if (p.suffix.lower() == ".mht") or (mtype == "message/rfc822") or (p.suffix.lower() == ".mhtml"): - # .mht parsing: try to parse parts and split into pages - with open(p.as_posix(), "rb") as f: - data = f.read() - try: - mht_pages = _parse_mht_bytes(data) - if mht_pages: - pages.extend(mht_pages) - else: - # fallback to raw read - content = _read_local_html_file(p.as_posix()) - pages.append({"id": p.name, "title": p.stem, "content": content}) - except Exception as e: - logger.exception("Failed to parse MHT file %s: %s", p, e) - content = _read_local_html_file(p.as_posix()) - pages.append({"id": p.name, "title": p.stem, "content": content}) - else: - content = _read_local_html_file(p.as_posix()) - pages.append({"id": p.name, "title": p.stem, "content": content}) - return {"type": "local", "pages": pages} - - # directory: search for .html/.htm/.mht/.mhtml - exts = (".html", ".htm", ".mht", ".mhtml") - for fp in sorted(p.rglob("*")): - if fp.suffix.lower() in exts and fp.is_file(): - try: - if fp.suffix.lower() in (".mht", ".mhtml"): - with open(fp.as_posix(), "rb") as f: - data = f.read() - mht_pages = _parse_mht_bytes(data) - if mht_pages: - for mp in mht_pages: - pages.append({"id": fp.as_posix() + "::" + mp.get("id", ""), "title": mp.get("title"), "content": mp.get("content")}) - continue - # else fall through to raw read - content = _read_local_html_file(fp.as_posix()) - except Exception as e: - logger.exception("Failed to read file %s: %s", fp, e) - continue - # try to extract title from HTML - m = re.search(r"(.*?)", content, flags=re.I | re.S) - title = m.group(1).strip() if m else fp.stem - pages.append({"id": fp.as_posix(), "title": title, "content": content}) - return {"type": "local", "pages": pages} - - -def _safe_filename(s: str) -> str: - # create a filesystem-safe filename - s = s or "page" - s = re.sub(r"[\\\/\:\*\?\"<>\|]", "_", s) - s = re.sub(r"\s+", "_", s) - return s[:180] - - -def _maybe_debug_save(pages: List[Dict[str, Any]], options: Optional[Dict[str, Any]] = None) -> None: - """If options contains 'debug_save_dir', save each page's HTML into that directory for inspection.""" - options = options or {} - outdir = options.get("debug_save_dir") - if not outdir: - return - outp = pathlib.Path(outdir) - try: - outp.mkdir(parents=True, exist_ok=True) - except Exception: - return - for idx, page in enumerate(pages): - try: - title = page.get("title") or f"page_{idx}" - fname = f"{idx:03d}_{_safe_filename(title)}.html" - fp = outp.joinpath(fname) - with open(fp, "w", encoding="utf-8") as f: - f.write(page.get("content") or "") - except Exception: - logger.exception("Failed to write debug page %s", page.get("id")) - - -# Small smoke-run when executed directly -if __name__ == "__main__": - import sys - if len(sys.argv) > 1: - path = sys.argv[1] - else: - path = r'C:\Users\zhouyr9\Documents\2025.mht' - - with open(path, "r", encoding='latin-1') as f: - content = f.read() - out = load_OneNote_document(path) - print(f"Found {len(out['pages'])} pages")