mirror of
https://github.com/tradecatlabs/vibe-coding-cn.git
synced 2026-08-19 05:48:04 +00:00
fix: prompts-library jsonl directory import
This commit is contained in:
@@ -10,6 +10,7 @@
|
|||||||
- Markdown -> Excel:把 Markdown 文档目录还原为 Excel 工作簿。
|
- Markdown -> Excel:把 Markdown 文档目录还原为 Excel 工作簿。
|
||||||
- Markdown -> JSONL:把 Markdown 提示词转换为 JSONL。
|
- Markdown -> JSONL:把 Markdown 提示词转换为 JSONL。
|
||||||
- JSONL -> Excel:把 JSONL 数据转换为 Excel。
|
- JSONL -> Excel:把 JSONL 数据转换为 Excel。
|
||||||
|
- JSONL 目录 -> Excel:把 Excel(JSONL) 导出的工作表级 JSONL 目录重新合并为 Excel。
|
||||||
- Excel(JSONL) -> JSONL:把内部 JSONL 格式的 Excel 按工作表拆成 JSONL 文件。
|
- Excel(JSONL) -> JSONL:把内部 JSONL 格式的 Excel 按工作表拆成 JSONL 文件。
|
||||||
|
|
||||||
## 目录结构
|
## 目录结构
|
||||||
@@ -58,6 +59,7 @@ python3 main.py --select "prompt_excel/example.xlsx" --mode excel2docs
|
|||||||
python3 main.py --select "prompt_docs/example" --mode docs2excel
|
python3 main.py --select "prompt_docs/example" --mode docs2excel
|
||||||
python3 main.py --select "prompt_docs/example" --mode docs2jsonl
|
python3 main.py --select "prompt_docs/example" --mode docs2jsonl
|
||||||
python3 main.py --select "prompt_jsonl/example.jsonl" --mode jsonl2excel
|
python3 main.py --select "prompt_jsonl/example.jsonl" --mode jsonl2excel
|
||||||
|
python3 main.py --select "prompt_jsonl/example_2025_1222_004537" --mode jsonl2excel
|
||||||
python3 main.py --select "prompt_excel/prompt_jsonl.xlsx" --mode jsonl_excel2jsonl
|
python3 main.py --select "prompt_excel/prompt_jsonl.xlsx" --mode jsonl_excel2jsonl
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ Unified controller for prompt-library conversions.
|
|||||||
3. Docs → JSONL : 将 Markdown 文档转换为 JSONL 格式(保留完整元信息)
|
3. Docs → JSONL : 将 Markdown 文档转换为 JSONL 格式(保留完整元信息)
|
||||||
4. JSONL → Excel : 将 JSONL 转换为 Excel(单元格存储 JSON 对象)
|
4. JSONL → Excel : 将 JSONL 转换为 Excel(单元格存储 JSON 对象)
|
||||||
5. Excel(JSONL) → JSONL : 将内部 JSONL 格式的 Excel 转换为 JSONL 目录(自动忽略"说明"工作表)
|
5. Excel(JSONL) → JSONL : 将内部 JSONL 格式的 Excel 转换为 JSONL 目录(自动忽略"说明"工作表)
|
||||||
|
6. JSONL 目录 → Excel : 将 Excel(JSONL) 导出的工作表级 JSONL 目录重新合并为 Excel
|
||||||
|
|
||||||
数据格式规范
|
数据格式规范
|
||||||
============
|
============
|
||||||
@@ -51,6 +52,7 @@ JSONL → Excel 单元格格式:
|
|||||||
- Docs→Excel: ./prompt_excel/prompt_excel_YYYY_MMDD_HHMMSS/rebuilt.xlsx
|
- Docs→Excel: ./prompt_excel/prompt_excel_YYYY_MMDD_HHMMSS/rebuilt.xlsx
|
||||||
- Docs→JSONL: ./prompt_jsonl/{docs_name}.jsonl
|
- Docs→JSONL: ./prompt_jsonl/{docs_name}.jsonl
|
||||||
- JSONL→Excel: ./prompt_excel/{jsonl_name}.xlsx
|
- JSONL→Excel: ./prompt_excel/{jsonl_name}.xlsx
|
||||||
|
- JSONL目录→Excel: ./prompt_excel/{jsonl_dir_name}.xlsx
|
||||||
- Excel(JSONL)→JSONL: ./prompt_jsonl/{excel_name}_{timestamp}/<sheet>.jsonl
|
- Excel(JSONL)→JSONL: ./prompt_jsonl/{excel_name}_{timestamp}/<sheet>.jsonl
|
||||||
|
|
||||||
使用示例
|
使用示例
|
||||||
@@ -69,6 +71,7 @@ JSONL → Excel 单元格格式:
|
|||||||
|
|
||||||
# JSONL → Excel
|
# JSONL → Excel
|
||||||
python3 main.py --select "prompt_jsonl/prompt_docs.jsonl"
|
python3 main.py --select "prompt_jsonl/prompt_docs.jsonl"
|
||||||
|
python3 main.py --select "prompt_jsonl/prompt_jsonl_2025_1222_004537" --mode jsonl2excel
|
||||||
|
|
||||||
# Excel(JSONL) → JSONL(自动检测或显式指定)
|
# Excel(JSONL) → JSONL(自动检测或显式指定)
|
||||||
python3 main.py --select "prompt_excel/prompt_jsonl.xlsx"
|
python3 main.py --select "prompt_excel/prompt_jsonl.xlsx"
|
||||||
@@ -108,7 +111,7 @@ except Exception: # pragma: no cover
|
|||||||
@dataclass
|
@dataclass
|
||||||
class Candidate:
|
class Candidate:
|
||||||
index: int
|
index: int
|
||||||
kind: str # "excel" | "docs" | "docs2jsonl" | "jsonl"
|
kind: str # "excel" | "docs" | "docs2jsonl" | "jsonl" | "jsonl_dir" | "jsonl_excel"
|
||||||
path: Path
|
path: Path
|
||||||
label: str
|
label: str
|
||||||
|
|
||||||
@@ -221,6 +224,20 @@ def list_jsonl_files(jsonl_dir: Path) -> List[Path]:
|
|||||||
return sorted([p for p in jsonl_dir.iterdir() if p.is_file() and p.suffix.lower() == ".jsonl"], key=lambda p: p.stat().st_mtime)
|
return sorted([p for p in jsonl_dir.iterdir() if p.is_file() and p.suffix.lower() == ".jsonl"], key=lambda p: p.stat().st_mtime)
|
||||||
|
|
||||||
|
|
||||||
|
def list_jsonl_dirs(jsonl_dir: Path) -> List[Path]:
|
||||||
|
"""列出由 Excel(JSONL)→JSONL 生成的工作表级 JSONL 目录。"""
|
||||||
|
if not jsonl_dir.exists():
|
||||||
|
return []
|
||||||
|
return sorted(
|
||||||
|
[
|
||||||
|
p
|
||||||
|
for p in jsonl_dir.iterdir()
|
||||||
|
if p.is_dir() and any(child.is_file() and child.suffix.lower() == ".jsonl" for child in p.iterdir())
|
||||||
|
],
|
||||||
|
key=lambda p: p.stat().st_mtime,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def is_jsonl_excel(excel_path: Path) -> bool:
|
def is_jsonl_excel(excel_path: Path) -> bool:
|
||||||
"""检测 Excel 是否为内部 JSONL 格式(单元格存储 JSON 对象)"""
|
"""检测 Excel 是否为内部 JSONL 格式(单元格存储 JSON 对象)"""
|
||||||
import json
|
import json
|
||||||
@@ -383,8 +400,21 @@ def run_jsonl_excel_to_jsonl(excel_path: Path, project_root: Path) -> int:
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
def run_jsonl_to_excel(jsonl_path: Path, project_root: Path) -> int:
|
def read_jsonl_records(jsonl_paths: Sequence[Path]) -> List[dict]:
|
||||||
"""Convert JSONL to Excel, each cell contains the full JSON object as string."""
|
import json
|
||||||
|
records = []
|
||||||
|
|
||||||
|
for jsonl_path in jsonl_paths:
|
||||||
|
with open(jsonl_path, 'r', encoding='utf-8') as f:
|
||||||
|
for line in f:
|
||||||
|
if line.strip():
|
||||||
|
records.append(json.loads(line))
|
||||||
|
|
||||||
|
return records
|
||||||
|
|
||||||
|
|
||||||
|
def write_records_to_excel(records: List[dict], output_file: Path) -> int:
|
||||||
|
"""将 JSONL 记录写回 Excel,单元格中只保留 title/content JSON。"""
|
||||||
import json
|
import json
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
try:
|
try:
|
||||||
@@ -392,56 +422,79 @@ def run_jsonl_to_excel(jsonl_path: Path, project_root: Path) -> int:
|
|||||||
except ImportError:
|
except ImportError:
|
||||||
print("❌ 需要 pandas: pip install pandas openpyxl")
|
print("❌ 需要 pandas: pip install pandas openpyxl")
|
||||||
return 1
|
return 1
|
||||||
|
|
||||||
records = []
|
|
||||||
with open(jsonl_path, 'r', encoding='utf-8') as f:
|
|
||||||
for line in f:
|
|
||||||
if line.strip():
|
|
||||||
records.append(json.loads(line))
|
|
||||||
|
|
||||||
if not records:
|
if not records:
|
||||||
print(f"❌ JSONL 文件为空: {jsonl_path}")
|
print("❌ JSONL 文件为空")
|
||||||
return 1
|
return 1
|
||||||
|
|
||||||
# category -> {row -> {col -> json_string}}
|
# category -> {row -> {col -> json_string}}
|
||||||
sheets_data: dict = defaultdict(lambda: defaultdict(dict))
|
sheets_data: dict = defaultdict(lambda: defaultdict(dict))
|
||||||
cat_id_map = {}
|
cat_id_map = {}
|
||||||
|
|
||||||
for r in records:
|
for r in records:
|
||||||
cat_name = r["category"]
|
cat_name = r["category"]
|
||||||
cat_id_map[r["category_id"]] = cat_name
|
cat_id_map[r["category_id"]] = cat_name
|
||||||
# 单元格内容只保留 title 和 content
|
# 单元格内容只保留 title 和 content
|
||||||
cell_data = {"title": r["title"], "content": r["content"]}
|
cell_data = {"title": r["title"], "content": r["content"]}
|
||||||
sheets_data[cat_name][r["row"]][r["col"]] = json.dumps(cell_data, ensure_ascii=False)
|
sheets_data[cat_name][r["row"]][r["col"]] = json.dumps(cell_data, ensure_ascii=False)
|
||||||
|
|
||||||
output_dir = project_root / "prompt_excel"
|
output_file.parent.mkdir(parents=True, exist_ok=True)
|
||||||
output_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
output_file = output_dir / f"{jsonl_path.stem}.xlsx"
|
|
||||||
|
|
||||||
sorted_cats = sorted(cat_id_map.items(), key=lambda x: x[0])
|
sorted_cats = sorted(cat_id_map.items(), key=lambda x: x[0])
|
||||||
|
|
||||||
with pd.ExcelWriter(output_file, engine='openpyxl') as writer:
|
with pd.ExcelWriter(output_file, engine='openpyxl') as writer:
|
||||||
for cat_id, cat_name in sorted_cats:
|
for cat_id, cat_name in sorted_cats:
|
||||||
row_data = sheets_data[cat_name]
|
row_data = sheets_data[cat_name]
|
||||||
if not row_data:
|
if not row_data:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
max_row = max(row_data.keys())
|
max_row = max(row_data.keys())
|
||||||
max_col = max(c for cols in row_data.values() for c in cols.keys())
|
max_col = max(c for cols in row_data.values() for c in cols.keys())
|
||||||
|
|
||||||
data = []
|
data = []
|
||||||
for row_idx in range(1, max_row + 1):
|
for row_idx in range(1, max_row + 1):
|
||||||
row_list = []
|
row_list = []
|
||||||
for col_idx in range(1, max_col + 1):
|
for col_idx in range(1, max_col + 1):
|
||||||
row_list.append(row_data.get(row_idx, {}).get(col_idx, ""))
|
row_list.append(row_data.get(row_idx, {}).get(col_idx, ""))
|
||||||
data.append(row_list)
|
data.append(row_list)
|
||||||
|
|
||||||
df = pd.DataFrame(data)
|
df = pd.DataFrame(data)
|
||||||
sheet_name = cat_name[:31]
|
sheet_name = cat_name[:31]
|
||||||
df.to_excel(writer, sheet_name=sheet_name, index=False, header=False)
|
df.to_excel(writer, sheet_name=sheet_name, index=False, header=False)
|
||||||
|
|
||||||
print(f"✅ JSONL→Excel OK: {jsonl_path.name} → {output_file.relative_to(project_root)} ({len(sorted_cats)} 个工作表)")
|
return len(sorted_cats)
|
||||||
return 0
|
|
||||||
|
|
||||||
|
def run_jsonl_to_excel(jsonl_path: Path, project_root: Path) -> int:
|
||||||
|
"""Convert one JSONL file to Excel, each cell contains the full JSON object as string."""
|
||||||
|
output_dir = project_root / "prompt_excel"
|
||||||
|
output_file = output_dir / f"{jsonl_path.stem}.xlsx"
|
||||||
|
sheet_count = write_records_to_excel(read_jsonl_records([jsonl_path]), output_file)
|
||||||
|
if sheet_count > 0:
|
||||||
|
print(f"✅ JSONL→Excel OK: {jsonl_path.name} → {output_file.relative_to(project_root)} ({sheet_count} 个工作表)")
|
||||||
|
return 0
|
||||||
|
return sheet_count
|
||||||
|
|
||||||
|
|
||||||
|
def run_jsonl_dir_to_excel(jsonl_dir: Path, project_root: Path) -> int:
|
||||||
|
"""Convert a directory of sheet-level JSONL files back to one Excel workbook."""
|
||||||
|
jsonl_paths = sorted(
|
||||||
|
[p for p in jsonl_dir.iterdir() if p.is_file() and p.suffix.lower() == ".jsonl"],
|
||||||
|
key=lambda p: p.name,
|
||||||
|
)
|
||||||
|
if not jsonl_paths:
|
||||||
|
print(f"❌ JSONL 目录中没有 .jsonl 文件: {jsonl_dir}")
|
||||||
|
return 1
|
||||||
|
|
||||||
|
output_dir = project_root / "prompt_excel"
|
||||||
|
output_file = output_dir / f"{jsonl_dir.name}.xlsx"
|
||||||
|
sheet_count = write_records_to_excel(read_jsonl_records(jsonl_paths), output_file)
|
||||||
|
if sheet_count > 0:
|
||||||
|
print(
|
||||||
|
f"✅ JSONL目录→Excel OK: {jsonl_dir.name} → "
|
||||||
|
f"{output_file.relative_to(project_root)} ({sheet_count} 个工作表 / {len(jsonl_paths)} 个 JSONL 文件)"
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
return sheet_count
|
||||||
|
|
||||||
|
|
||||||
def build_candidates(project_root: Path, excel_dir: Path, docs_dir: Path) -> List[Candidate]:
|
def build_candidates(project_root: Path, excel_dir: Path, docs_dir: Path) -> List[Candidate]:
|
||||||
@@ -469,6 +522,10 @@ def build_candidates(project_root: Path, excel_dir: Path, docs_dir: Path) -> Lis
|
|||||||
label = f"{path.name}"
|
label = f"{path.name}"
|
||||||
candidates.append(Candidate(index=idx, kind="jsonl", path=path, label=label))
|
candidates.append(Candidate(index=idx, kind="jsonl", path=path, label=label))
|
||||||
idx += 1
|
idx += 1
|
||||||
|
for path in list_jsonl_dirs(jsonl_dir):
|
||||||
|
label = f"{path.name}/"
|
||||||
|
candidates.append(Candidate(index=idx, kind="jsonl_dir", path=path, label=label))
|
||||||
|
idx += 1
|
||||||
return candidates
|
return candidates
|
||||||
|
|
||||||
|
|
||||||
@@ -508,7 +565,7 @@ def select_interactively(candidates: Sequence[Candidate]) -> Optional[Candidate]
|
|||||||
table.add_column("编号", style="bold yellow", justify="right", width=4)
|
table.add_column("编号", style="bold yellow", justify="right", width=4)
|
||||||
table.add_column("类型", style="magenta", width=16)
|
table.add_column("类型", style="magenta", width=16)
|
||||||
table.add_column("路径/名称", style="white")
|
table.add_column("路径/名称", style="white")
|
||||||
kind_labels = {"excel": "Excel→Docs", "docs": "Docs→Excel", "docs2jsonl": "Docs→JSONL", "jsonl": "JSONL→Excel", "jsonl_excel": "Excel(JSONL)→JSONL"}
|
kind_labels = {"excel": "Excel→Docs", "docs": "Docs→Excel", "docs2jsonl": "Docs→JSONL", "jsonl": "JSONL→Excel", "jsonl_excel": "Excel(JSONL)→JSONL", "jsonl_dir": "JSONL目录→Excel"}
|
||||||
for c in candidates:
|
for c in candidates:
|
||||||
table.add_row(str(c.index), kind_labels.get(c.kind, c.kind), c.label)
|
table.add_row(str(c.index), kind_labels.get(c.kind, c.kind), c.label)
|
||||||
|
|
||||||
@@ -530,7 +587,7 @@ def select_interactively(candidates: Sequence[Candidate]) -> Optional[Candidate]
|
|||||||
console.print("[red]编号不存在,请重试[/red]")
|
console.print("[red]编号不存在,请重试[/red]")
|
||||||
|
|
||||||
# Plain fallback
|
# Plain fallback
|
||||||
kind_labels = {"excel": "Excel→Docs", "docs": "Docs→Excel", "docs2jsonl": "Docs→JSONL", "jsonl": "JSONL→Excel", "jsonl_excel": "Excel(JSONL)→JSONL"}
|
kind_labels = {"excel": "Excel→Docs", "docs": "Docs→Excel", "docs2jsonl": "Docs→JSONL", "jsonl": "JSONL→Excel", "jsonl_excel": "Excel(JSONL)→JSONL", "jsonl_dir": "JSONL目录→Excel"}
|
||||||
print("请选择一个源进行转换:")
|
print("请选择一个源进行转换:")
|
||||||
for c in candidates:
|
for c in candidates:
|
||||||
print(f" {c.index:2d}. [{kind_labels.get(c.kind, c.kind)}] {c.label}")
|
print(f" {c.index:2d}. [{kind_labels.get(c.kind, c.kind)}] {c.label}")
|
||||||
@@ -596,6 +653,11 @@ def main() -> int:
|
|||||||
if selected.is_file() and selected.suffix.lower() == ".jsonl":
|
if selected.is_file() and selected.suffix.lower() == ".jsonl":
|
||||||
return run_jsonl_to_excel(selected, repo_root)
|
return run_jsonl_to_excel(selected, repo_root)
|
||||||
if selected.is_dir():
|
if selected.is_dir():
|
||||||
|
if args.mode == "jsonl2excel" or (
|
||||||
|
selected.parent == (repo_root / "prompt_jsonl").resolve()
|
||||||
|
and any(p.is_file() and p.suffix.lower() == ".jsonl" for p in selected.iterdir())
|
||||||
|
):
|
||||||
|
return run_jsonl_dir_to_excel(selected, repo_root)
|
||||||
# Check mode or default to docs2excel
|
# Check mode or default to docs2excel
|
||||||
if args.mode == "docs2jsonl":
|
if args.mode == "docs2jsonl":
|
||||||
return run_docs_to_jsonl(selected, repo_root)
|
return run_docs_to_jsonl(selected, repo_root)
|
||||||
@@ -616,6 +678,8 @@ def main() -> int:
|
|||||||
return run_docs_to_jsonl(chosen.path, repo_root)
|
return run_docs_to_jsonl(chosen.path, repo_root)
|
||||||
elif chosen.kind == "jsonl":
|
elif chosen.kind == "jsonl":
|
||||||
return run_jsonl_to_excel(chosen.path, repo_root)
|
return run_jsonl_to_excel(chosen.path, repo_root)
|
||||||
|
elif chosen.kind == "jsonl_dir":
|
||||||
|
return run_jsonl_dir_to_excel(chosen.path, repo_root)
|
||||||
else:
|
else:
|
||||||
return run_start_convert(start_convert, mode="docs2excel", project_root=repo_root, select_path=chosen.path, docs_dir=docs_dir)
|
return run_start_convert(start_convert, mode="docs2excel", project_root=repo_root, select_path=chosen.path, docs_dir=docs_dir)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user