基于Flask实现ipynb自动转HTML并构建可搜索只读索引
实现自动IPython Notebook索引与搜索的Flask方案
作为经常和科研笔记打交道的人,我完全理解你不想手动重复转换的痛点。下面是一套完整的解决方案,让Flask自动帮你搞定笔记转换、索引和搜索,全程无需手动干预:
核心思路
我们要构建一个Flask服务,它能自动扫描指定目录下的.ipynb文件,只转换新增或修改过的笔记(避免重复劳动),同时生成可全文搜索的索引页面。整个流程分为三个核心模块:自动转换逻辑、HTML模板集成、轻量级搜索功能。
步骤1:安装依赖
先把需要的工具包装齐,除了Flask,我们用nbconvert做格式转换,whoosh实现轻量级全文搜索(比数据库更适合科研场景的轻量需求):
pip install flask nbconvert python-dotenv whoosh
步骤2:项目结构规划
推荐这样的目录结构,清晰易维护:
notebook-indexer/ ├── .env # 配置文件(指定笔记目录、缓存目录等) ├── app.py # Flask主程序 ├── templates/ # HTML模板 │ ├── base.html # 基础模板(导航、搜索框) │ ├── index.html # 索引列表页面 │ └── notebook.html # 单个笔记的渲染模板 ├── static/ # 静态资源(自定义CSS) │ └── style.css └── cache/ # 存放转换后的HTML(自动创建) └── ...
步骤3:自动转换逻辑实现
在app.py里先写基础配置和转换函数,确保只处理更新过的文件:
from flask import Flask, render_template, request, send_from_directory from nbconvert import HTMLExporter from pathlib import Path from datetime import datetime import os from dotenv import load_dotenv from whoosh.index import create_in, open_dir from whoosh.fields import Schema, TEXT, ID, DATETIME from whoosh.qparser import QueryParser # 加载配置(从.env文件读取) load_dotenv() NOTEBOOK_DIR = Path(os.getenv("NOTEBOOK_DIR", "./notebooks")) CACHE_DIR = Path(os.getenv("CACHE_DIR", "./cache")) INDEX_DIR = Path(os.getenv("INDEX_DIR", "./search_index")) # 自动创建必要的目录 CACHE_DIR.mkdir(exist_ok=True) INDEX_DIR.mkdir(exist_ok=True) app = Flask(__name__) def convert_notebook(notebook_path): """将单个ipynb文件转换为HTML,保存到缓存目录""" exporter = HTMLExporter() exporter.template_name = "classic" # 也可以替换为自定义模板 body, resources = exporter.from_filename(notebook_path) # 生成和原文件路径一致的缓存路径 relative_path = notebook_path.relative_to(NOTEBOOK_DIR) cache_path = CACHE_DIR / relative_path.with_suffix(".html") cache_path.parent.mkdir(exist_ok=True) # 写入HTML文件 with open(cache_path, "w", encoding="utf-8") as f: f.write(body) return cache_path def sync_notebooks(): """扫描目录,同步所有更新过的ipynb文件到缓存""" updated_notebooks = [] # 遍历所有ipynb文件,跳过隐藏文件/目录 for notebook_path in NOTEBOOK_DIR.rglob("*.ipynb"): if any(part.startswith(".") for part in notebook_path.parts): continue # 检查缓存是否存在,或原文件是否更新 relative_path = notebook_path.relative_to(NOTEBOOK_DIR) cache_path = CACHE_DIR / relative_path.with_suffix(".html") if not cache_path.exists() or notebook_path.stat().st_mtime > cache_path.stat().st_mtime: convert_notebook(notebook_path) updated_notebooks.append(notebook_path) # 更新搜索索引 update_search_index(updated_notebooks) return len(updated_notebooks)
步骤4:实现全文搜索功能
用Whoosh构建轻量级搜索索引,支持按笔记内容和标题检索:
# 定义搜索索引的结构 schema = Schema( path=ID(stored=True, unique=True), title=TEXT(stored=True), content=TEXT(stored=True), modified=DATETIME(stored=True) ) def update_search_index(updated_notebooks): """只更新新增/修改笔记的搜索索引""" # 打开或创建索引 if INDEX_DIR.exists() and len(list(INDEX_DIR.iterdir())) > 0: ix = open_dir(INDEX_DIR) else: ix = create_in(INDEX_DIR, schema) writer = ix.writer() for notebook_path in updated_notebooks: # 提取笔记标题(优先用文件名,也可以改逻辑读第一个markdown单元格) title = notebook_path.stem # 读取笔记内容用于搜索(包含代码和markdown) import nbformat nb = nbformat.read(notebook_path, as_version=4) content = "\n".join([cell["source"] for cell in nb.cells if cell["cell_type"] in ["markdown", "code"]]) modified = datetime.fromtimestamp(notebook_path.stat().st_mtime) # 删除旧索引(如果存在),再添加新索引 writer.delete_by_term("path", str(notebook_path)) writer.add_document( path=str(notebook_path), title=title, content=content, modified=modified ) writer.commit()
步骤5:Flask路由与模板集成
写路由来展示索引、搜索结果和单个笔记,同时在启动时自动同步一次:
# 启动Flask时自动同步所有笔记 sync_notebooks() @app.route("/") def index(): # 获取所有笔记的元信息,按修改时间排序 notebooks = [] for notebook_path in NOTEBOOK_DIR.rglob("*.ipynb"): if any(part.startswith(".") for part in notebook_path.parts): continue relative_path = notebook_path.relative_to(NOTEBOOK_DIR) notebooks.append({ "title": notebook_path.stem, "path": str(relative_path), "modified": datetime.fromtimestamp(notebook_path.stat().st_mtime).strftime("%Y-%m-%d %H:%M") }) notebooks.sort(key=lambda x: x["modified"], reverse=True) return render_template("index.html", notebooks=notebooks) @app.route("/notebook/<path:path>") def show_notebook(path): notebook_path = NOTEBOOK_DIR / path if not notebook_path.exists() or notebook_path.suffix != ".ipynb": return "Notebook not found", 404 # 确保缓存是最新的(防止用户直接访问时笔记已更新) cache_path = CACHE_DIR / Path(path).with_suffix(".html") if not cache_path.exists() or notebook_path.stat().st_mtime > cache_path.stat().st_mtime: convert_notebook(notebook_path) # 读取缓存的HTML并嵌入模板 with open(cache_path, "r", encoding="utf-8") as f: notebook_html = f.read() return render_template("notebook.html", title=notebook_path.stem, content=notebook_html) @app.route("/search") def search(): query_str = request.args.get("q", "") results = [] if query_str: ix = open_dir(INDEX_DIR) with ix.searcher() as searcher: query = QueryParser("content", ix.schema).parse(query_str) hits = searcher.search(query, limit=20) for hit in hits: notebook_path = Path(hit["path"]) relative_path = notebook_path.relative_to(NOTEBOOK_DIR) results.append({ "title": hit["title"], "path": str(relative_path), "modified": hit["modified"].strftime("%Y-%m-%d %H:%M") }) return render_template("index.html", notebooks=results, query=query_str)
步骤6:编写HTML模板
base.html(基础模板,包含导航和搜索框)
<!DOCTYPE html> <html lang="en"> <head> <meta charset="UTF-8"> <meta name="viewport" content="width=device-width, initial-scale=1.0"> <title>{% block title %}Research Notebook Index{% endblock %}</title> <link rel="stylesheet" href="{{ url_for('static', filename='style.css') }}"> </head> <body> <nav> <h1>Lab Notebook Index</h1> <form action="/search" method="get"> <input type="text" name="q" placeholder="Search notebooks..." value="{{ query or '' }}"> <button type="submit">Search</button> </form> <a href="/">Back to Full Index</a> </nav> <div class="content"> {% block content %}{% endblock %} </div> </body> </html>
index.html(索引列表页面)
{% extends "base.html" %} {% block title %}Notebook Index{% endblock %} {% block content %} {% if query %} <h2>Search Results for "{{ query }}"</h2> {% else %} <h2>All Lab Notebooks</h2> {% endif %} <ul class="notebook-list"> {% for notebook in notebooks %} <li> <a href="{{ url_for('show_notebook', path=notebook.path) }}">{{ notebook.title }}</a> <span class="modified">{{ notebook.modified }}</span> </li> {% endfor %} </ul> {% endblock %}
notebook.html(单个笔记展示页面)
{% extends "base.html" %} {% block title %}{{ title }}{% endblock %} {% block content %} <h2>{{ title }}</h2> <div class="notebook-content"> {{ content|safe }} </div> {% endblock %}
步骤7:添加定时自动同步(可选)
如果不想每次重启Flask才同步,可以用APScheduler定期检查更新:
pip install apscheduler
然后在app.py里添加:
from apscheduler.schedulers.background import BackgroundScheduler # 每10分钟自动同步一次笔记 scheduler = BackgroundScheduler() scheduler.add_job(sync_notebooks, 'interval', minutes=10) scheduler.start() # 关闭Flask时停止调度器 @app.teardown_appcontext def shutdown_scheduler(exception=None): scheduler.shutdown()
运行与使用
- 在
.env文件里配置你的笔记目录:
NOTEBOOK_DIR=/path/to/your/research/notebooks
- 启动Flask服务:
flask run
- 访问
http://localhost:5000就能看到笔记索引,搜索功能实时可用,新增或修改的笔记会自动被转换并加入索引。
额外优化建议
- 自定义样式:修改
static/style.css调整页面布局,比如优化代码块的可读性、索引列表的排版。 - 智能标题提取:可以修改代码,读取笔记本的第一个markdown单元格作为标题,比文件名更直观。
- 团队权限控制:如果需要限制访问,可添加
Flask-Login做简单的身份验证,适合科研团队内部使用。
内容的提问来源于stack exchange,提问作者Jonathan Wheeler
相关产品推荐
相关产品推荐

