From b2c195c365b6a2a83f262b20fb1b2f15afae1dcb Mon Sep 17 00:00:00 2001 From: Martin Tazl Date: Tue, 21 Jul 2026 16:32:49 +0200 Subject: [PATCH] Initial project structure --- .gitignore | 37 ++ LICENSE | 0 README.md | 0 docs/architecture.md | 259 +++++++++++ docs/data-models.md | 275 +++++++++++ docs/experiments.md | 0 docs/pipeline.md | 428 ++++++++++++++++++ prompts/decisions.md | 0 prompts/facts.md | 0 prompts/positions.md | 0 prompts/questions.md | 0 prompts/technical.md | 0 prompts/todos.md | 0 pyproject.toml | 0 src/meeting_lab/__init__.py | 0 src/meeting_lab/chunking/__init__.py | 0 src/meeting_lab/chunking/chunk_transcript.py | 0 src/meeting_lab/consolidation/__init__.py | 0 src/meeting_lab/consolidation/merge_topics.py | 0 src/meeting_lab/extraction/__init__.py | 0 .../extraction/extract_decisions.py | 0 src/meeting_lab/extraction/extract_facts.py | 0 .../extraction/extract_positions.py | 0 .../extraction/extract_questions.py | 0 .../extraction/extract_technical.py | 0 src/meeting_lab/extraction/extract_todos.py | 0 src/meeting_lab/io/__init__.py | 0 src/meeting_lab/io/files.py | 0 src/meeting_lab/io/json_io.py | 0 src/meeting_lab/llm/__init__.py | 0 src/meeting_lab/llm/models.py | 0 src/meeting_lab/llm/ollama.py | 0 src/meeting_lab/llm/prompts.py | 0 src/meeting_lab/models/__init__.py | 0 src/meeting_lab/models/chunk.py | 0 src/meeting_lab/models/extraction.py | 0 src/meeting_lab/models/meeting.py | 0 src/meeting_lab/models/topic.py | 0 src/meeting_lab/normalization/__init__.py | 0 .../normalization/normalize_transcript.py | 0 src/meeting_lab/normalization/rules.py | 0 src/meeting_lab/protocol/__init__.py | 0 src/meeting_lab/protocol/build_protocol.py | 0 src/meeting_lab/segmentation/__init__.py | 0 .../segmentation/segment_topics.py | 0 45 files changed, 999 insertions(+) create mode 100644 .gitignore create mode 100644 LICENSE create mode 100644 README.md create mode 100644 docs/architecture.md create mode 100644 docs/data-models.md create mode 100644 docs/experiments.md create mode 100644 docs/pipeline.md create mode 100644 prompts/decisions.md create mode 100644 prompts/facts.md create mode 100644 prompts/positions.md create mode 100644 prompts/questions.md create mode 100644 prompts/technical.md create mode 100644 prompts/todos.md create mode 100644 pyproject.toml create mode 100644 src/meeting_lab/__init__.py create mode 100644 src/meeting_lab/chunking/__init__.py create mode 100644 src/meeting_lab/chunking/chunk_transcript.py create mode 100644 src/meeting_lab/consolidation/__init__.py create mode 100644 src/meeting_lab/consolidation/merge_topics.py create mode 100644 src/meeting_lab/extraction/__init__.py create mode 100644 src/meeting_lab/extraction/extract_decisions.py create mode 100644 src/meeting_lab/extraction/extract_facts.py create mode 100644 src/meeting_lab/extraction/extract_positions.py create mode 100644 src/meeting_lab/extraction/extract_questions.py create mode 100644 src/meeting_lab/extraction/extract_technical.py create mode 100644 src/meeting_lab/extraction/extract_todos.py create mode 100644 src/meeting_lab/io/__init__.py create mode 100644 src/meeting_lab/io/files.py create mode 100644 src/meeting_lab/io/json_io.py create mode 100644 src/meeting_lab/llm/__init__.py create mode 100644 src/meeting_lab/llm/models.py create mode 100644 src/meeting_lab/llm/ollama.py create mode 100644 src/meeting_lab/llm/prompts.py create mode 100644 src/meeting_lab/models/__init__.py create mode 100644 src/meeting_lab/models/chunk.py create mode 100644 src/meeting_lab/models/extraction.py create mode 100644 src/meeting_lab/models/meeting.py create mode 100644 src/meeting_lab/models/topic.py create mode 100644 src/meeting_lab/normalization/__init__.py create mode 100644 src/meeting_lab/normalization/normalize_transcript.py create mode 100644 src/meeting_lab/normalization/rules.py create mode 100644 src/meeting_lab/protocol/__init__.py create mode 100644 src/meeting_lab/protocol/build_protocol.py create mode 100644 src/meeting_lab/segmentation/__init__.py create mode 100644 src/meeting_lab/segmentation/segment_topics.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..aaffe1d --- /dev/null +++ b/.gitignore @@ -0,0 +1,37 @@ +# Python +__pycache__/ +*.py[cod] +*.so + +# Virtual Environment +.venv/ +venv/ + +# Build +build/ +dist/ +*.egg-info/ + +# IDE +.vscode/ +.idea/ + +# macOS +.DS_Store + +# Test +.pytest_cache/ +.coverage +htmlcov/ + +# Ruff / mypy +.ruff_cache/ +.mypy_cache/ + +# Experiment Outputs +experiments/**/output/ +experiments/**/results/ + +# Lokale Meetings (niemals versionieren) +meeting_data/ +recordings/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..e69de29 diff --git a/README.md b/README.md new file mode 100644 index 0000000..e69de29 diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..69ee691 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,259 @@ +# Architecture + +## Purpose + +The **Meeting Lab** is an experimental environment for developing and evaluating methods to extract structured knowledge from real meeting transcripts. + +Its purpose is not to build a complete meeting assistant, but to answer a single question: + +> **How can knowledge be extracted from real discussions as reliably as possible?** + +Successful approaches will later be integrated into the Meeting Assistant project. + +--- + +# Design Goals + +The architecture follows a small set of guiding principles. + +## Modular Pipeline + +Complex problems are divided into small, well-defined processing steps. + +Each module has exactly one responsibility. + +## Deterministic where possible + +Tasks that can be solved reliably without an LLM should use deterministic algorithms. + +Examples include: + +- transcript normalization +- whitespace cleanup +- duplicate removal +- chunk generation + +LLMs are only used where semantic understanding is required. + +## Preserve Information + +The pipeline should never remove or rewrite information unless it is certain that the content is merely noise. + +Losing information is considered worse than keeping harmless redundancy. + +## Explainable Results + +Every processing step should be understandable. + +Intermediate results should remain inspectable throughout the pipeline. + +## Reproducible Experiments + +Experiments must be repeatable. + +Given the same input, prompt, model and parameters, another developer should be able to reproduce the result. + +## Local First + +The complete pipeline should run locally. + +Cloud services may be supported in the future but are not a design requirement. + +--- + +# Core Idea + +Traditional meeting summarization attempts to solve everything in one step. + +```text +Transcript + ↓ +LLM + ↓ +Summary +``` + +Real discussions do not work that way. + +Topics are introduced, interrupted, resumed later, expanded, questioned and finally concluded. + +Instead of building a better summarizer, the Meeting Lab develops a **Discussion Analyzer**. + +The analyzer gradually transforms an unstructured discussion into structured knowledge. + +--- + +# High-Level Pipeline + +```text +Transcript + ↓ +Normalization + ↓ +Discussion Blocks + ↓ +Technical Chunking + ↓ +Topic Segmentation + ↓ +Specialized Extraction + ↓ +Consolidation + ↓ +Structured Meeting Data + ↓ +Protocol Generation +``` + +Each stage solves one clearly defined problem. + +No module should perform multiple semantic tasks simultaneously. + +--- + +# Module Overview + +The current architecture consists of the following processing stages. + +## normalization/ + +Deterministic transcript cleanup. + +Responsibilities: + +- remove filler words +- remove immediate repetitions +- whitespace cleanup +- generate change log + +--- + +## chunking/ + +Creates model-sized chunks. + +Chunking is purely technical. + +It does **not** recognize discussion topics. + +--- + +## segmentation/ + +Identifies discussion topics. + +Responsibilities: + +- detect topic start +- detect topic end +- detect topic switches +- recognize resumed topics + +This is the next major development milestone. + +--- + +## extraction/ + +Contains specialized LLM modules. + +Planned extractors include: + +- facts +- questions +- positions +- decisions +- todos +- technical information + +Each extractor has exactly one task and one prompt. + +--- + +## consolidation/ + +Merges information extracted from multiple discussion segments. + +Typical responsibilities: + +- merge duplicates +- combine partial information +- distinguish positions from decisions +- detect contradictions + +--- + +## protocol/ + +Generates human-readable output from structured meeting data. + +Protocol generation never invents information. + +It only reformulates the analysis results. + +--- + +# Repository Layout + +```text +meeting-lab/ +│ +├── src/ +├── prompts/ +├── experiments/ +├── samples/ +├── tests/ +└── docs/ +``` + +Additional documentation is intentionally split into focused documents. + +Examples: + +- pipeline.md +- segmentation.md +- prompts.md +- experiments.md + +The architecture document only describes the overall system. + +--- + +# Current State + +Implemented: + +- Transcript normalization +- Technical chunk generation +- Experimental LLM-based information extraction + +The current extraction step still performs multiple tasks simultaneously. + +This was sufficient as a proof of concept but does not reflect the intended long-term architecture. + +--- + +# Next Milestone + +The next development step is the implementation of **topic segmentation**. + +Its only responsibility is to identify the thematic structure of a discussion. + +It should answer questions such as: + +- Where does a topic begin? +- Where does it end? +- When does another topic start? +- When is an earlier topic resumed? + +No facts, decisions or todos should be extracted at this stage. + +Only after reliable topic segmentation has been achieved will the specialized extraction modules be implemented. + +--- + +# Guiding Principle + +The Meeting Lab assumes that the greatest improvement in transcript quality will not come from increasingly powerful language models. + +Instead, quality is expected to emerge from a pipeline that decomposes a complex problem into many small, clearly defined and independently testable processing steps. \ No newline at end of file diff --git a/docs/data-models.md b/docs/data-models.md new file mode 100644 index 0000000..653b09d --- /dev/null +++ b/docs/data-models.md @@ -0,0 +1,275 @@ +# Data Models + +## Purpose + +This document describes the logical data structures exchanged between the pipeline stages of the Meeting Lab. + +The goal is **not** to define a final database schema. + +Instead, these models represent stable interfaces between processing modules. + +Models should evolve only when required by new functionality. + +--- + +# Design Principles + +## Keep Models Small + +Only include fields that are currently required. + +Avoid speculative attributes. + +Bad: + +```json +{ + "priority": "...", + "confidence": 0.93, + "risk": "...", + "category": "...", + "importance": "...", + "status": "..." +} +``` + +Good: + +```json +{ + "text": "...", + "owner": "..." +} +``` + +New fields can always be added later. + +--- + +## Preserve Information + +Models should preserve information rather than interpret it. + +Interpretation belongs to processing modules. + +--- + +## Stable Interfaces + +Modules communicate only through documented data models. + +A module must never depend on another module's internal implementation. + +--- + +# Transcript + +Represents the complete meeting transcript. + +Example + +```json +{ + "meeting_id": "meeting_001", + "language": "en", + "blocks": [] +} +``` + +--- + +# Discussion Block + +The discussion block is the fundamental processing unit. + +```json +{ + "block_id": 42, + "speaker": "Speaker A", + "start": 351.2, + "end": 367.8, + "text": "..." +} +``` + +Required fields + +- block_id +- text + +Optional fields + +- speaker +- timestamps + +--- + +# Chunk + +Technical processing unit. + +```json +{ + "chunk_id": 3, + "blocks": [ + 40, + 41, + 42 + ] +} +``` + +Chunks are implementation details. + +They never represent discussion topics. + +--- + +# Topic + +Represents one discussion topic. + +```json +{ + "topic_id": "topic_003", + "title": "Ventilation", + "segments": [] +} +``` + +--- + +# Topic Segment + +A continuous part of a topic. + +```json +{ + "start_block": 40, + "end_block": 152 +} +``` + +One topic may contain multiple segments. + +--- + +# Fact + +```json +{ + "text": "..." +} +``` + +--- + +# Question + +```json +{ + "text": "..." +} +``` + +--- + +# Position + +```json +{ + "text": "...", + "speaker": "..." +} +``` + +--- + +# Decision + +```json +{ + "text": "..." +} +``` + +--- + +# Todo + +```json +{ + "text": "...", + "owner": "..." +} +``` + +Owner remains empty if unknown. + +--- + +# Technical Detail + +```json +{ + "text": "..." +} +``` + +--- + +# Topic Result + +After extraction, every topic contains the collected information. + +```json +{ + "topic_id": "topic_003", + "title": "Ventilation", + + "segments": [], + + "facts": [], + "questions": [], + "positions": [], + "decisions": [], + "todos": [], + "technical_details": [] +} +``` + +This object represents the main output of the analysis pipeline. + +--- + +# Meeting Result + +The complete structured meeting. + +```json +{ + "meeting_id": "meeting_001", + + "topics": [] +} +``` + +Protocol generation operates exclusively on this structure. + +--- + +# Future Extensions + +Possible future additions include: + +- confidence values +- evidence references +- source blocks +- priorities +- deadlines +- status tracking +- semantic relationships + +These fields will only be introduced when they provide measurable benefits. + +The Meeting Lab intentionally avoids designing an overly complex schema in advance. \ No newline at end of file diff --git a/docs/experiments.md b/docs/experiments.md new file mode 100644 index 0000000..e69de29 diff --git a/docs/pipeline.md b/docs/pipeline.md new file mode 100644 index 0000000..6be0251 --- /dev/null +++ b/docs/pipeline.md @@ -0,0 +1,428 @@ +# Pipeline + +## Purpose + +This document describes the processing pipeline of the Meeting Lab. + +Unlike `architecture.md`, which describes the overall system, this document focuses on the individual processing stages, their responsibilities and the data flowing between them. + +The guiding principle is simple: + +> **Each processing stage has exactly one responsibility.** + +--- + +# Pipeline Overview + +```text +Whisper Transcript + │ + ▼ +Normalization + │ + ▼ +Discussion Blocks + │ + ▼ +Technical Chunking + │ + ▼ +Topic Segmentation + │ + ▼ +Specialized Extraction + │ + ▼ +Consolidation + │ + ▼ +Structured Meeting + │ + ▼ +Protocol Generation +``` + +Each stage receives a well-defined input and produces a well-defined output. + +--- + +# Stage 1 – Normalization + +## Purpose + +Remove transcription artifacts without changing the meaning of the discussion. + +## Input + +Raw transcript generated by Whisper. + +## Output + +Normalized transcript. + +Change log containing every modification. + +## Processing Type + +Deterministic + +## Responsibilities + +- Remove filler words +- Remove immediate duplicate words +- Remove immediate duplicate short phrases +- Normalize whitespace +- Preserve all semantic content + +## Must Not + +- Rephrase text +- Summarize +- Interpret statements +- Correct factual content + +## Current Status + +Implemented + +--- + +# Stage 2 – Discussion Blocks + +## Purpose + +Convert the transcript into stable processing units. + +Discussion blocks are the smallest semantic unit used throughout the pipeline. + +## Input + +Normalized transcript. + +## Output + +Ordered list of discussion blocks. + +Example: + +```json +{ + "block_id": 42, + "speaker": "A", + "start": 351.2, + "end": 367.8, + "text": "..." +} +``` + +## Processing Type + +Deterministic + +## Responsibilities + +- Create stable identifiers +- Preserve ordering +- Preserve timestamps +- Preserve speaker information where available + +## Current Status + +Planned + +--- + +# Stage 3 – Technical Chunking + +## Purpose + +Split large meetings into model-sized chunks. + +Chunking exists only because language models have limited context windows. + +## Input + +Discussion blocks. + +## Output + +Chunk manifest and chunk files. + +## Processing Type + +Deterministic + +## Responsibilities + +- Respect block boundaries +- Keep chunk size below model limits +- Optionally create overlapping context + +## Must Not + +- Detect discussion topics +- Merge discussion content +- Interpret meaning + +## Current Status + +Implemented + +--- + +# Stage 4 – Topic Segmentation + +## Purpose + +Identify the thematic structure of the discussion. + +This is considered the central research problem of the Meeting Lab. + +## Input + +Discussion blocks or technical chunks. + +## Output + +Topics consisting of one or more discussion segments. + +Example: + +```json +{ + "topic": "Ventilation", + "segments": [ + { + "start_block": 40, + "end_block": 152 + }, + { + "start_block": 1618, + "end_block": 1697 + } + ] +} +``` + +## Processing Type + +LLM + +## Responsibilities + +- Detect topic start +- Detect topic end +- Detect topic changes +- Detect resumed topics +- Associate discussion blocks with topics + +## Must Not + +- Extract facts +- Detect todos +- Generate summaries + +## Current Status + +Planned + +--- + +# Stage 5 – Specialized Extraction + +## Purpose + +Extract one specific type of information from each topic. + +Every extractor performs exactly one task. + +## Planned Extractors + +```text +extract_facts.py +extract_questions.py +extract_positions.py +extract_decisions.py +extract_todos.py +extract_technical.py +``` + +Each extractor has: + +- one prompt +- one responsibility +- one output schema + +## Processing Type + +LLM + +## Current Status + +Prototype exists as a combined extractor. + +--- + +# Stage 6 – Consolidation + +## Purpose + +Merge analysis results originating from different discussion segments. + +## Input + +Extraction results. + +## Output + +Unified topic representation. + +## Responsibilities + +- Merge duplicates +- Merge complementary information +- Preserve contradictions +- Separate positions from decisions +- Combine related todos + +## Processing Type + +Hybrid + +Deterministic wherever possible. + +LLM support only if necessary. + +## Current Status + +Planned + +--- + +# Stage 7 – Structured Meeting + +## Purpose + +Produce a complete machine-readable representation of the meeting. + +This is the primary output of the analysis pipeline. + +Example: + +```json +{ + "topics": [ + { + "title": "...", + "facts": [], + "questions": [], + "positions": [], + "decisions": [], + "todos": [] + } + ] +} +``` + +The exact schema will evolve during development. + +## Current Status + +Planned + +--- + +# Stage 8 – Protocol Generation + +## Purpose + +Generate human-readable documents from structured meeting data. + +Possible outputs include: + +- Full protocol +- Executive summary +- Action list +- Decision log +- Technical report + +Protocol generation never performs additional analysis. + +It only transforms existing structured information into readable text. + +## Processing Type + +LLM + +## Current Status + +Planned + +--- + +# Data Flow + +Each stage consumes only the output of the previous stage. + +```text +Stage N + │ +Structured Output + │ + ▼ +Stage N + 1 +``` + +Intermediate results remain available for inspection, testing and experimentation. + +--- + +# Guiding Principles + +Every pipeline stage should satisfy the following rules. + +## Single Responsibility + +One module. + +One task. + +## Explicit Input + +Every stage expects a clearly defined input format. + +## Explicit Output + +Every stage produces a clearly defined output format. + +## Independent Evaluation + +Each stage should be testable without executing the entire pipeline. + +## Replaceable Components + +A processing stage may be replaced by another implementation as long as it preserves the same interface. + +--- + +# Current Development Roadmap + +```text +✔ Normalization + +⬜ Discussion Blocks + +✔ Technical Chunking + +⬜ Topic Segmentation + +⬜ Specialized Extraction + +⬜ Consolidation + +⬜ Structured Meeting + +⬜ Protocol Generation +``` + +The immediate development focus is **Topic Segmentation**, as it provides the semantic structure on which all subsequent processing stages depend. \ No newline at end of file diff --git a/prompts/decisions.md b/prompts/decisions.md new file mode 100644 index 0000000..e69de29 diff --git a/prompts/facts.md b/prompts/facts.md new file mode 100644 index 0000000..e69de29 diff --git a/prompts/positions.md b/prompts/positions.md new file mode 100644 index 0000000..e69de29 diff --git a/prompts/questions.md b/prompts/questions.md new file mode 100644 index 0000000..e69de29 diff --git a/prompts/technical.md b/prompts/technical.md new file mode 100644 index 0000000..e69de29 diff --git a/prompts/todos.md b/prompts/todos.md new file mode 100644 index 0000000..e69de29 diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/__init__.py b/src/meeting_lab/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/chunking/__init__.py b/src/meeting_lab/chunking/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/chunking/chunk_transcript.py b/src/meeting_lab/chunking/chunk_transcript.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/consolidation/__init__.py b/src/meeting_lab/consolidation/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/consolidation/merge_topics.py b/src/meeting_lab/consolidation/merge_topics.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/__init__.py b/src/meeting_lab/extraction/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_decisions.py b/src/meeting_lab/extraction/extract_decisions.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_facts.py b/src/meeting_lab/extraction/extract_facts.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_positions.py b/src/meeting_lab/extraction/extract_positions.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_questions.py b/src/meeting_lab/extraction/extract_questions.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_technical.py b/src/meeting_lab/extraction/extract_technical.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/extraction/extract_todos.py b/src/meeting_lab/extraction/extract_todos.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/io/__init__.py b/src/meeting_lab/io/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/io/files.py b/src/meeting_lab/io/files.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/io/json_io.py b/src/meeting_lab/io/json_io.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/llm/__init__.py b/src/meeting_lab/llm/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/llm/models.py b/src/meeting_lab/llm/models.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/llm/ollama.py b/src/meeting_lab/llm/ollama.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/llm/prompts.py b/src/meeting_lab/llm/prompts.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/models/__init__.py b/src/meeting_lab/models/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/models/chunk.py b/src/meeting_lab/models/chunk.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/models/extraction.py b/src/meeting_lab/models/extraction.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/models/meeting.py b/src/meeting_lab/models/meeting.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/models/topic.py b/src/meeting_lab/models/topic.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/normalization/__init__.py b/src/meeting_lab/normalization/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/normalization/normalize_transcript.py b/src/meeting_lab/normalization/normalize_transcript.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/normalization/rules.py b/src/meeting_lab/normalization/rules.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/protocol/__init__.py b/src/meeting_lab/protocol/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/protocol/build_protocol.py b/src/meeting_lab/protocol/build_protocol.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/segmentation/__init__.py b/src/meeting_lab/segmentation/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/meeting_lab/segmentation/segment_topics.py b/src/meeting_lab/segmentation/segment_topics.py new file mode 100644 index 0000000..e69de29