-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathcoord.py
More file actions
210 lines (169 loc) · 6.52 KB
/
Copy pathcoord.py
File metadata and controls
210 lines (169 loc) · 6.52 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
#!/usr/bin/env python3
"""
MegaMind — Extract, structure, and track valuable content.
Usage:
python coord.py <URL> Extract content from a URL
python coord.py --paste <source_type> Paste content manually
python coord.py --list Show all extractions
python coord.py --list --status TODO Filter by status
python coord.py --status <num> <status> Update entry status
"""
import argparse
import sys
from datetime import datetime, timezone
from extractors import get_extractor
from extractors.detector import SourceType
from extractors.base import ExtractionResult
from processors.ai_processor import process_extraction
from outputs.formatter import format_document, generate_filename
from outputs.index import add_to_index, update_status, list_entries
from outputs.storage import save_extraction
def run_pipeline(url: str) -> dict:
"""Reusable extraction pipeline. Returns structured result dict.
Used by both the CLI and the Discord bot.
"""
# 1. Detect source and get extractor
extractor, source_type = get_extractor(url)
# 2. Extract raw content
result = extractor.extract(url)
# Keep the actual fetched text locally for later source review. The public
# repository receives only the processed note, never raw transcript files.
from source_evidence import save_snapshot
filename = generate_filename(result)
save_snapshot(filename, result.raw_content, result.url,
result.metadata.get("extraction_method", "scrape"))
# 3. Process through AI
processed = process_extraction(result)
# 4. Format final document
document = format_document(result, processed)
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
# 5. Save to storage
saved = save_extraction(filename, document)
# 6. Update index
add_to_index(result, processed, filename, date_str)
return {
"title": result.title,
"url": result.url,
"source_type": source_type.value,
"processed": processed,
"document": document,
"filename": filename,
"saved_to": saved,
"date": date_str,
"metadata": result.metadata,
}
def extract_url(url: str) -> None:
"""CLI extraction pipeline for a given URL."""
print(f"\n MegaMind")
print(f" {'='*40}")
print(f" Extracting: {url}")
result = run_pipeline(url)
print(f" Source: {result['source_type']}")
print(f" Title: {result['title']}")
for location, path in result["saved_to"].items():
print(f" -> {location}: {path}")
print(f" Index updated.")
print(f"\n Done! Extraction saved as: {result['filename']}")
print(f" {'='*40}\n")
def paste_content(source_type_str: str) -> None:
"""Handle manual paste mode for any source type."""
type_map = {
"youtube": SourceType.YOUTUBE,
"twitter": SourceType.TWITTER,
"x": SourceType.TWITTER,
"github": SourceType.GITHUB,
"article": SourceType.ARTICLE,
}
source_type = type_map.get(source_type_str.lower())
if not source_type:
print(f"Unknown source type: {source_type_str}")
print(f"Valid types: {', '.join(type_map.keys())}")
sys.exit(1)
print(f"\n MegaMind — Manual Paste Mode")
print(f" {'='*40}")
print(f" Source type: {source_type.value}")
url = input("\n URL (or press Enter if none): ").strip() or "N/A"
title = input(" Title: ").strip() or "Untitled"
print(f"\n Paste content below (press Enter twice when done):\n")
lines = []
empty_count = 0
try:
while True:
line = input()
if line == "":
empty_count += 1
if empty_count >= 2:
break
lines.append(line)
else:
empty_count = 0
lines.append(line)
except EOFError:
pass
content = "\n".join(lines).strip()
if not content:
print("No content provided. Exiting.")
sys.exit(1)
result = ExtractionResult(
title=title,
url=url,
source_type=source_type.value,
raw_content=content,
metadata={"extraction_method": "manual_paste"},
)
filename = generate_filename(result)
from source_evidence import save_snapshot
save_snapshot(filename, result.raw_content, result.url, "manual_paste")
print(f" Processing with AI...")
processed = process_extraction(result)
document = format_document(result, processed)
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
saved = save_extraction(filename, document)
for location, path in saved.items():
print(f" -> {location}: {path}")
add_to_index(result, processed, filename, date_str)
print(f"\n Done! Extraction saved as: {filename}")
print(f" {'='*40}\n")
def main():
parser = argparse.ArgumentParser(
description="MegaMind — Extract, structure, and track valuable content.",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""\
Examples:
python coord.py https://youtube.com/watch?v=abc123
python coord.py https://x.com/user/status/123456
python coord.py https://github.com/owner/repo
python coord.py https://example.com/article
python coord.py --paste youtube
python coord.py --list
python coord.py --list --status TODO
python coord.py --status 3 "In Progress"
""",
)
parser.add_argument("url", nargs="?", help="URL to extract content from")
parser.add_argument("--paste", metavar="TYPE", help="Manual paste mode (youtube, twitter, github, article)")
parser.add_argument("--list", action="store_true", help="List all extractions from the index")
parser.add_argument("--status", nargs=2, metavar=("NUM", "STATUS"),
help='Update status of entry NUM (e.g., --status 3 "In Progress")')
parser.add_argument("--filter", metavar="STATUS", help="Filter --list by status (Backlog, TODO, In Progress, Done)")
args = parser.parse_args()
if args.list:
print(list_entries(args.filter))
return
if args.status:
entry_num = int(args.status[0])
new_status = args.status[1]
if update_status(entry_num, new_status):
print(f" Entry #{entry_num} updated to: {new_status}")
else:
print(f" Entry #{entry_num} not found.")
return
if args.paste:
paste_content(args.paste)
return
if args.url:
extract_url(args.url)
return
parser.print_help()
if __name__ == "__main__":
main()