File size: 2,236 Bytes
eb1c19a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
"""LLM Output Length Prediction Module.

This module provides comprehensive feature extraction and prediction
capabilities for estimating LLM response lengths from input prompts.

Features are organized into 5 categories:
1. Text Statistics - Length, vocabulary, compression metrics
2. Structural - Questions, lists, code blocks, formatting
3. Semantic - Task type, domain, complexity indicators
4. Embedding - Neural embeddings and similarity scores
5. Meta - Model settings, historical patterns

Example:
    from headroom.prediction import PromptFeatureExtractor, extract_features

    # Full extractor (with embeddings)
    extractor = PromptFeatureExtractor(use_embeddings=True)
    features = extractor.extract("What is machine learning?", model="gpt-4o")

    # Quick extraction (no embeddings)
    features = extract_features("Explain quantum computing")

    # Get ML-ready vector
    vector = features.to_vector()
    names = features.feature_names()

Install full dependencies:
    pip install headroom[prediction]

This installs:
    - sentence-transformers (for embedding features)
    - spacy (for NER, optional)
"""

from .feature_extractor import (
    ComplexityLevel,
    DomainType,
    EmbeddingExtractor,
    EmbeddingFeatures,
    MetaExtractor,
    MetaFeatures,
    # Main extractor
    PromptFeatureExtractor,
    # Feature dataclasses
    PromptFeatures,
    PromptFormat,
    SemanticExtractor,
    SemanticFeatures,
    StructuralExtractor,
    StructuralFeatures,
    # Enums
    TaskType,
    # Individual extractors
    TextStatisticsExtractor,
    TextStatisticsFeatures,
    # Utility functions
    extract_features,
    get_feature_vector,
)

__all__ = [
    # Main extractor
    "PromptFeatureExtractor",
    # Individual extractors
    "TextStatisticsExtractor",
    "StructuralExtractor",
    "SemanticExtractor",
    "EmbeddingExtractor",
    "MetaExtractor",
    # Feature dataclasses
    "PromptFeatures",
    "TextStatisticsFeatures",
    "StructuralFeatures",
    "SemanticFeatures",
    "EmbeddingFeatures",
    "MetaFeatures",
    # Enums
    "TaskType",
    "DomainType",
    "ComplexityLevel",
    "PromptFormat",
    # Utility functions
    "extract_features",
    "get_feature_vector",
]