From 212ec3a9fc7322f068b09533a17c62b5f1404623 Mon Sep 17 00:00:00 2001 From: Maziyar Panahi Date: Sun, 19 Jul 2026 19:04:26 +0200 Subject: [PATCH] docs: add task-oriented cookbook index --- docs/cookbook.md | 65 ++++++++++++++++++++++++++ docs/examples.md | 4 ++ mkdocs.yml | 1 + tests/unit/test_docs_cookbook_links.py | 52 +++++++++++++++++++++ 4 files changed, 122 insertions(+) create mode 100644 docs/cookbook.md create mode 100644 tests/unit/test_docs_cookbook_links.py diff --git a/docs/cookbook.md b/docs/cookbook.md new file mode 100644 index 000000000..de39db58d --- /dev/null +++ b/docs/cookbook.md @@ -0,0 +1,65 @@ +# Cookbook + +Start with the task you need to complete, then open the linked script, +notebook, or copy-ready documentation. The examples use synthetic data unless +their own instructions say otherwise. Use only authorized data, and keep raw +PHI out of logs, notebook outputs, and shared artifacts. + +For a feature-oriented tour with inline snippets, see +[Examples & Copy/Paste Recipes](./examples.md). + +## PII detection and de-identification + +| Goal | Open this asset | +| --- | --- | +| Learn the complete PII workflow, from detection through de-identification | [`PII_Detection_Complete_Guide.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/PII_Detection_Complete_Guide.ipynb) | +| De-identify a list or CSV, batch a folder, or run a reversible replacement round trip | [`Deidentification_Cookbook.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/Deidentification_Cookbook.ipynb) | +| Batch PII extraction or de-identification in Python | [`pii_batch_processing.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/pii_batch_processing.py) | +| Compare PII models on the same synthetic text | [`pii_model_comparison.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/pii_model_comparison.py) | +| Evaluate multilingual PII detection and model selection | [`Multilingual_PII_Detection_Guide.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/Multilingual_PII_Detection_Guide.ipynb) and [`pii_multilingual_new_languages.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/pii_multilingual_new_languages.py) | +| Generate consistent, seeded, locale-aware replacement values | [`obfuscation_demo.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/obfuscation_demo.py) | +| Try de-identification in a small Gradio app | [`gradio_deid_app.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/gradio_deid_app.py) | +| Use one Privacy Filter API across MLX and PyTorch | [`privacy_filter_unified.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_unified.py) | +| Compare Privacy Filter model families in an interactive app | [`privacy_filter_book/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_book/README.md) and [`privacy_filter_book/app.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_book/app.py) | +| Run a focused Privacy Filter studio | [`privacy_filter_studio/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_studio/README.md) and [`privacy_filter_studio/app.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_studio/app.py) | +| Compare multilingual Privacy Filter results side by side | [`privacy_filter_multilingual_studio/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_multilingual_studio/README.md) and [`privacy_filter_multilingual_studio/app.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_filter_multilingual_studio/app.py) | +| Exercise the public PII API with a bundled synthetic fixture | [`datasets_walkthrough.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/datasets_walkthrough.py) | + +## Clinical NER and model exploration + +| Goal | Open this asset | +| --- | --- | +| Learn installation, registry exploration, and a first `analyze_text` call | [`getting_started.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/getting_started.ipynb) | +| Compare disease, pharmaceutical, and oncology NER families | [`clinical_ner_families.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/clinical_ner_families.py) | +| Explore zero-shot NER and custom entity labels | [`ZeroShot_NER_Tour.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/ZeroShot_NER_Tour.ipynb) | +| Run GLiNER zero-shot NER through MLX | [`mlx_gliner_zero_shot_ner.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/mlx_gliner_zero_shot_ner.py) | +| Run a converted token-classification model through MLX | [`mlx_token_classification_ner.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/mlx_token_classification_ner.py) | +| Export registry operations as agent-framework tool definitions | [`agent_tools_quickstart.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/agent_tools_quickstart.py) | +| Render clinical extraction results as a dataframe | [`clinical_extraction_dataframe_api.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/clinical_extraction_dataframe_api.py) | + +## Batching and tokenization + +| Goal | Open this asset | +| --- | --- | +| Segment long text, batch sentences, and align predictions to paragraphs | [`Sentence_Detection_Batching.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/Sentence_Detection_Batching.ipynb) | +| Learn medical-aware tokenization interactively | [`Medical_Tokenizer_Demo.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/Medical_Tokenizer_Demo.ipynb) | +| Benchmark medical tokenization performance | [`Medical_Tokenizer_Benchmark.ipynb`](https://github.com/maziyarpanahi/openmed/blob/master/examples/notebooks/Medical_Tokenizer_Benchmark.ipynb) | +| Align custom tokens back to encoder outputs | [`custom_tokenize_alignment.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/custom_tokenizer/custom_tokenize_alignment.py) | +| Compare encoder output with medical remapping on and off | [`compare_medical_remap.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/custom_tokenizer/compare_medical_remap.py) | +| Evaluate tokenizer strategies on a clinical corpus | [`eval_tokenization_comparison.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/custom_tokenizer/eval_tokenization_comparison.py) | +| Review the custom-tokenizer setup and commands | [`custom_tokenizer/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/custom_tokenizer/README.md) | + +## Services, pipelines, and safeguards + +| Goal | Open this asset | +| --- | --- | +| Start the REST service and call health, analysis, PII, and de-identification endpoints | [REST API Recipes](./rest-recipes.md) | +| Redact before an external model call and restore identifiers locally | [`privacy_gateway_quickstart.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/privacy_gateway_quickstart.py) | +| Redact selected fields in NDJSON log events | [`log-redaction/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/log-redaction/README.md) | +| De-identify warehouse columns with dbt macros | [`dbt-deidentify/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/dbt-deidentify/README.md) | +| De-identify Spark Structured Streaming micro-batches | [`spark-streaming/README.md`](https://github.com/maziyarpanahi/openmed/blob/master/examples/spark-streaming/README.md) | +| Add a de-identification screen to a React Native app | [`react-native/RedactScreen.tsx`](https://github.com/maziyarpanahi/openmed/blob/master/examples/react-native/RedactScreen.tsx) | +| Review policy, audit, and release-evidence safeguards offline | [`v16_policy_audit_release_gates.py`](https://github.com/maziyarpanahi/openmed/blob/master/examples/v16_policy_audit_release_gates.py) | + +Grounding, FHIR, and multimodal workflows are maintained in their dedicated +guides and are intentionally outside this cookbook's scope. diff --git a/docs/examples.md b/docs/examples.md index 97a30066f..63d44bc1e 100644 --- a/docs/examples.md +++ b/docs/examples.md @@ -4,6 +4,10 @@ This page curates the most useful samples already in the repository so you can jump straight to runnable notebooks or scripts. The v1.6, v1.7, and v1.8 examples use synthetic data and are safe to run during release review. +If you already know the task you want to complete, use the +[Cookbook](./cookbook.md) to jump directly to the matching script, notebook, or +copy-ready recipe. + ## Notebooks (`examples/notebooks/`) | Notebook | Highlights | diff --git a/mkdocs.yml b/mkdocs.yml index 4c88f4c67..0be7bf3c6 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -43,6 +43,7 @@ nav: - Performance Profiling: profiling.md - Zero-shot NER How-to: zero-shot-howto.md - Zero-shot Toolkit: zero-shot-ner.md + - Cookbook: cookbook.md - Examples & Copy/Paste Recipes: examples.md - Document Adapter Provenance: document-adapter-provenance.md - EPUB Extraction: multimodal/epub-extraction.md diff --git a/tests/unit/test_docs_cookbook_links.py b/tests/unit/test_docs_cookbook_links.py new file mode 100644 index 000000000..a6fa1404d --- /dev/null +++ b/tests/unit/test_docs_cookbook_links.py @@ -0,0 +1,52 @@ +"""Keep the task-oriented cookbook linked to real repository assets.""" + +from __future__ import annotations + +import re +from pathlib import Path +from urllib.parse import unquote + +ROOT = Path(__file__).resolve().parents[2] +COOKBOOK = ROOT / "docs" / "cookbook.md" +EXAMPLES_DOC = ROOT / "docs" / "examples.md" +MKDOCS = ROOT / "mkdocs.yml" + +MARKDOWN_LINK = re.compile(r"\[[^\]]+\]\(([^)]+)\)") +SOURCE_PREFIX = "https://github.com/maziyarpanahi/openmed/blob/master/" +REQUIRED_ASSETS = { + "examples/clinical_ner_families.py", + "examples/notebooks/Deidentification_Cookbook.ipynb", + "examples/pii_batch_processing.py", + "examples/pii_model_comparison.py", +} + + +def _linked_example_paths() -> list[str]: + text = COOKBOOK.read_text(encoding="utf-8") + return [ + unquote(target.removeprefix(SOURCE_PREFIX).split("#", 1)[0]) + for target in MARKDOWN_LINK.findall(text) + if target.startswith(f"{SOURCE_PREFIX}examples/") + ] + + +def test_every_linked_example_asset_exists() -> None: + paths = _linked_example_paths() + + assert paths, "cookbook must link at least one repository example" + missing = [ + relative_path for relative_path in paths if not (ROOT / relative_path).is_file() + ] + assert not missing, f"cookbook links missing example assets: {missing}" + + +def test_cookbook_covers_required_assets_and_rest_recipes() -> None: + text = COOKBOOK.read_text(encoding="utf-8") + + assert REQUIRED_ASSETS <= set(_linked_example_paths()) + assert "[REST API Recipes](./rest-recipes.md)" in text + + +def test_cookbook_is_in_nav_and_cross_linked_from_examples() -> None: + assert "Cookbook: cookbook.md" in MKDOCS.read_text(encoding="utf-8") + assert "[Cookbook](./cookbook.md)" in EXAMPLES_DOC.read_text(encoding="utf-8")