Files
awesome-quant/parse.py
T

96 lines
2.3 KiB
Python
Raw Normal View History

2022-03-27 21:11:25 -03:00
import os
import re
import pandas as pd
from threading import Thread
from github import Github
# using an access token
g = Github(os.environ['GITHUB_ACCESS_TOKEN'])
2022-03-30 06:11:03 -03:00
2022-03-27 21:11:25 -03:00
def extract_repo(url):
2022-03-30 06:11:03 -03:00
reu = re.compile('^https://github.com/([\w-]+/[-\w\.]+)$')
m = reu.match(url)
if m:
return m.group(1)
else:
return ''
2022-03-27 21:11:25 -03:00
def get_last_commit(repo):
2022-03-30 06:11:03 -03:00
try:
if repo:
r = g.get_repo(repo)
cs = r.get_commits()
return cs[0].commit.author.date.strftime('%Y-%m-%d')
else:
return ''
except:
print('ERROR' + repo)
return 'error'
2022-03-27 21:11:25 -03:00
class Project(Thread):
2022-03-30 06:11:03 -03:00
def __init__(self, match, section):
super().__init__()
self._match = match
self.regs = None
self._section = section
def run(self):
m = self._match
is_github = 'github.com' in m.group(2)
is_cran = 'cran.r-project.org' in m.group(2)
repo = extract_repo(m.group(2))
last_commit = get_last_commit(repo)
self.regs = dict(
project=m.group(1),
section=self._section,
last_commit=last_commit,
url=m.group(2),
description=m.group(3),
github=is_github,
cran=is_cran,
repo=repo
)
2022-03-27 21:11:25 -03:00
projects = []
with open('README.md', 'r', encoding='utf8') as f:
2022-03-30 06:11:03 -03:00
ret = re.compile('^(#+) (.*)$')
rex = re.compile('^\s*- \[(.*)\]\((.*)\) - (.*)$')
m_titles = []
last_head_level = 0
for line in f:
m = rex.match(line)
if m:
p = Project(m, ' > '.join(m_titles[1:]))
p.start()
projects.append(p)
2022-03-27 21:11:25 -03:00
else:
2022-03-30 06:11:03 -03:00
m = ret.match(line)
if m:
hrs = m.group(1)
if len(hrs) > last_head_level:
m_titles.append(m.group(2))
else:
for n in range(last_head_level - len(hrs) + 1):
m_titles.pop()
m_titles.append(m.group(2))
last_head_level = len(hrs)
2022-03-27 21:11:25 -03:00
while True:
2022-03-30 06:11:03 -03:00
checks = [not p.is_alive() for p in projects]
if all(checks):
break
2022-03-27 21:11:25 -03:00
projects = [p.regs for p in projects]
df = pd.DataFrame(projects)
df.to_csv('projects.csv', index=False)
df.to_markdown('projects.md', index=False)