Coverage for website.py: 100%

303 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-10-01 05:56 +0000

1# This file is part of Vallenato.fr. 

2# 

3# Vallenato.fr is free software: you can redistribute it and/or modify 

4# it under the terms of the GNU Affero General Public License as published by 

5# the Free Software Foundation, either version 3 of the License, or 

6# (at your option) any later version. 

7# 

8# Vallenato.fr is distributed in the hope that it will be useful, 

9# but WITHOUT ANY WARRANTY; without even the implied warranty of 

10# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the 

11# GNU Affero General Public License for more details. 

12# 

13# You should have received a copy of the GNU Affero General Public License 

14# along with Vallenato.fr. If not, see <http://www.gnu.org/licenses/>. 

15 

16import datetime 

17import json 

18import logging 

19import os 

20import re 

21import shutil 

22import sys 

23 

24from sitemap import generator 

25from slugify import slugify 

26 

27from youtube import ( 

28 HttpError, 

29 yt_get_authenticated_service, 

30 yt_get_my_uploads_list, 

31 yt_list_my_uploaded_videos, 

32) 

33 

34logger = logging.getLogger(__name__) 

35 

36# File that can contain the data downloaded from YouTube 

37UPLOADED_VIDEOS_DUMP_FILE = "data/uploaded_videos_dump.json" 

38# File containing the list of videos that have hardcoded locations 

39LOCATION_SPECIAL_CASES_FILE = "data/location_special_cases.json" 

40# File containing the already-identified latitude/longitude 

41GEOLOCATIONS_FILE = "data/geolocations.json" 

42# Output file used for the website 

43WEBSITE_DATA_FILE = "../website/src/data.js" 

44# Sitemap file 

45SITEMAP_FILE = "../website/prod/sitemap.xml" 

46# Version of the external libraries 

47LEAFLET_VERSION = "1.9.4" 

48BOOTSTRAP_VERSION = "4.6.2" 

49JQUERY_VERSION = "3.7.1" 

50BOOTSTRAP_TOGGLE_VERSION = "3.6.1" 

51 

52 

53def get_dumped_uploaded_videos(dump_file): 

54 uploaded_videos = [] 

55 # Used a previously dumped file if it exists, to bypass the network transactions 

56 if os.path.exists(dump_file): 

57 with open(dump_file) as in_file: 

58 uploaded_videos = json.load(in_file) 

59 return uploaded_videos 

60 

61 

62def save_uploaded_videos(uploaded_videos, dump_file): 

63 with open(dump_file, "w") as out_file: 

64 json.dump(uploaded_videos, out_file, sort_keys=True, indent=2) 

65 

66 

67def determine_videos_slug(uploaded_videos): 

68 logger.debug("Determining each video's slug...") 

69 for vid in uploaded_videos: 

70 vid["slug"] = slugify(vid["title"]).replace("-desde-", "-") 

71 return uploaded_videos 

72 

73 

74def get_uploaded_videos(args, dump_file): 

75 uploaded_videos = get_dumped_uploaded_videos(dump_file) 

76 if not uploaded_videos: 

77 youtube = yt_get_authenticated_service(args) 

78 # Get the list of videos uploaded to YouTube 

79 try: 

80 uploads_playlist_id = yt_get_my_uploads_list(youtube) 

81 if uploads_playlist_id: 

82 uploaded_videos = yt_list_my_uploaded_videos( 

83 uploads_playlist_id, youtube 

84 ) 

85 logger.debug(f"Uploaded videos: {uploaded_videos}") 

86 else: 

87 logger.info("There is no uploaded videos playlist for this user.") 

88 except HttpError as e: 

89 logger.debug(f"An HTTP error {e.resp.status} occurred:\n{e.content}") 

90 logger.critical("Exiting...") 

91 sys.exit(19) 

92 # Create a slug for each video (to be used for the website URLs) 

93 uploaded_videos = determine_videos_slug(uploaded_videos) 

94 if args.dump_uploaded_videos: 

95 save_uploaded_videos(uploaded_videos, dump_file) 

96 return uploaded_videos 

97 

98 

99def identify_locations_names(uploaded_videos, location_special_cases_file, dump_file): 

100 logger.debug("Identify each video's location name") 

101 with open(location_special_cases_file) as in_file: 

102 special_cases = json.load(in_file) 

103 locations = {} 

104 incomplete_locations = False 

105 for vid in uploaded_videos: 

106 vid["location"] = identify_single_location_name(vid, special_cases) 

107 if not vid["location"]: 

108 incomplete_locations = True 

109 elif vid["location"] not in locations: 

110 locations[vid["location"]] = {"latitude": None, "longitude": None} 

111 if incomplete_locations: 

112 # The script is going to exit, to prevent unnecessary downloading from 

113 # YouTube again, save the downloaded information regardless of the 

114 # --dump_uploaded_videos parameter 

115 logger.warning( 

116 f"Dumping the list of uploaded videos from YouTube to the '{dump_file}' file, so as not to have to download it again after you have edited the '{location_special_cases_file}' file." 

117 ) 

118 save_uploaded_videos(uploaded_videos, dump_file) 

119 logger.critical( 

120 f"Please add the new/missing location to the file '{location_special_cases_file}'. Exiting..." 

121 ) 

122 sys.exit(20) 

123 logger.info(f"Found {len(locations)} different location name.") 

124 return (uploaded_videos, locations) 

125 

126 

127def identify_single_location_name(vid, special_cases): 

128 location = None 

129 if vid["id"] in special_cases: 

130 location = special_cases[vid["id"]] 

131 logger.debug("Video {}, location '{}'".format(vid["id"], location)) 

132 else: 

133 for search_string in (", desde ", ", cerca de "): 

134 loc_index = vid["title"].find(search_string) 

135 if loc_index > 0: 

136 location = vid["title"][loc_index + len(search_string) :] 

137 logger.debug("Video {}, location '{}'".format(vid["id"], location)) 

138 break 

139 

140 # Each video should now have a location identified. If not, this will end the script. 

141 if not location: 

142 logger.critical( 

143 "No Location found for {}, '{}'".format(vid["id"], vid["title"]) 

144 ) 

145 return location 

146 

147 

148def determine_geolocation(locations, geolocations_file): 

149 logger.debug(f"Searching geolocation for {len(locations)} locations...") 

150 # Load the list of saved geolocations 

151 with open(geolocations_file) as in_file: 

152 geolocations = json.load(in_file) 

153 incomplete_geolocations = 0 

154 for l in locations: 

155 if ( 

156 l in geolocations 

157 and geolocations[l]["latitude"] 

158 and geolocations[l]["longitude"] 

159 ): 

160 logger.debug(f"Geolocation found for {l}") 

161 locations[l]["latitude"] = geolocations[l]["latitude"] 

162 locations[l]["longitude"] = geolocations[l]["longitude"] 

163 else: 

164 logger.critical(f"No geolocation found for {l}.") 

165 # TODO: Search and suggest a geolocation 

166 geolocations[l] = {"latitude": None, "longitude": None} 

167 incomplete_geolocations += 1 

168 

169 if incomplete_geolocations > 0: 

170 # Save the geolocations_file file with the placeholders for the unknown latitude and longitude 

171 with open(geolocations_file, "w") as out_file: 

172 json.dump(geolocations, out_file, sort_keys=True, indent=2) 

173 logger.critical( 

174 f"Please add the {incomplete_geolocations} new/missing unknown latitude and longitude to the file '{geolocations_file}'. Exiting..." 

175 ) 

176 sys.exit(21) 

177 

178 logger.info(f"Found geolocation information for the {len(locations)} locations.") 

179 return locations 

180 

181 

182def add_videos_to_locations_array(uploaded_videos, locations): 

183 logger.debug("Adding videos in each location array...") 

184 for vid in uploaded_videos: 

185 if not "videos" in locations[vid["location"]]: 

186 locations[vid["location"]]["videos"] = [] 

187 locations[vid["location"]]["videos"].append(vid) 

188 return locations 

189 

190 

191def determine_locations_slug(locations): 

192 logger.debug("Determining each location's slug...") 

193 for loc in locations: 

194 locations[loc]["slug"] = slugify(loc) 

195 return locations 

196 

197 

198def save_website_data(locations, website_data_file): 

199 logger.debug("Save the updated dynamic data") 

200 json_content = json.dumps(locations, sort_keys=True, indent=2) 

201 # Make it JS (and not just JSON) for direct use in the HTML document 

202 js_content = f"var locations = {json_content};" 

203 with open(website_data_file, "w") as out_file: 

204 out_file.write(js_content) 

205 

206 

207def load_locations_from_website_data_file(website_data_file): 

208 # Used by --no-fetch: rebuild the website from what's already committed 

209 # to website/src/data.js instead of hitting the YouTube API, e.g. to 

210 # build the production Docker image without live OAuth credentials. 

211 logger.debug(f"Loading existing locations from {website_data_file} (--no-fetch)") 

212 with open(website_data_file) as in_file: 

213 # Remove the JS bits to keep only the JSON content 

214 return json.loads(in_file.read()[16:-1]) 

215 

216 

217def flatten_videos_from_locations(locations): 

218 uploaded_videos = [] 

219 for location in locations.values(): 

220 uploaded_videos.extend(location["videos"]) 

221 uploaded_videos.sort(key=lambda v: v["publishedAt"], reverse=True) 

222 return uploaded_videos 

223 

224 

225def ignored_files_in_prod(adir, filenames): 

226 ignored_files = [] 

227 if "../website/src" == adir: 

228 ignored_files = [ 

229 f"bootstrap-{BOOTSTRAP_VERSION}-dist", 

230 f"bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}", 

231 f"jquery-{JQUERY_VERSION}.slim.min.js", 

232 "leaflet", 

233 ] 

234 if "../website/src/aprender" == adir: 

235 ignored_files = ["temp", "videos"] 

236 return [filename for filename in filenames if filename in ignored_files] 

237 

238 

239def get_stats(locations, uploaded_videos): 

240 num_videos = len(uploaded_videos) 

241 

242 songs = [] 

243 skipped_titles = ["Vallenato at Epic", "La Guaneña navideña"] 

244 for v in uploaded_videos: 

245 song = v["title"].split(",")[0] 

246 if song not in songs and song not in skipped_titles: 

247 songs.append(song) 

248 num_songs = len(songs) 

249 

250 num_places = len(locations) 

251 

252 countries = [] 

253 for l in locations: 

254 country = l.split(",")[-1] 

255 if country not in countries: 

256 countries.append(country) 

257 num_countries = len(countries) 

258 

259 navidad_2017 = datetime.date(2017, 12, 25) 

260 today = datetime.date.today() 

261 years = today.year - navidad_2017.year 

262 if today.month == 12: # December 

263 duration_since_navidad_2017 = f"{years} años" 

264 elif today.month == 1: # January 

265 duration_since_navidad_2017 = f"{years - 1} años" 

266 else: 

267 duration_since_navidad_2017 = f"{years - 1} años y {today.month} meses" 

268 

269 stats = f"El Vallenatero Francés les presenta {num_videos} videos de {num_songs} canciones tocadas en {num_places} lugares de {num_countries} paises. El empezo a aprender el Acordeón Vallenato en la Navidad 2017 (hace mas o menos {duration_since_navidad_2017})." 

270 

271 return stats 

272 

273 

274def generate_website(locations, uploaded_videos): 

275 logger.debug("Generate the production website files") 

276 input_src_folder = "../website/src" 

277 output_prod_folder = "../website/prod" 

278 # The 2 index files in / and /aprender, 404 

279 num_html_pages_created = 3 

280 

281 # Delete the previous production output folder (if existing) 

282 if os.path.exists(output_prod_folder): 

283 shutil.rmtree(output_prod_folder) 

284 

285 # Update statistics 

286 stats = get_stats(locations, uploaded_videos) 

287 index_src_file = f"{input_src_folder}/index.html" 

288 with open(index_src_file, "r") as file: 

289 index_data = file.read() 

290 index_data = re.sub( 

291 '<div id="stats">.*</div>', f'<div id="stats">{stats}</div>', index_data 

292 ) 

293 with open(index_src_file, "w") as file: 

294 file.write(index_data) 

295 

296 # Update the values accordingly for prod 

297 # Main difference between development (src) and production websites: 

298 # - src contains a full copy of the leaflet, Bootstrap and jQuery libraries 

299 # - prod uses CDNs 

300 

301 # Copy src to prod folder, ignoring the files and folder replaced by CDNs in prod 

302 # The videos are also not copied, as we're going to hard-link them 

303 shutil.copytree(input_src_folder, output_prod_folder, ignore=ignored_files_in_prod) 

304 

305 # Create hard links for the videos in the prod folder 

306 # (hard links can only be created for files, need to recreate the folder structure) 

307 os.mkdir(f"{output_prod_folder}/aprender/videos") 

308 # website/src/aprender/videos/ is gitignored (and .dockerignore'd) - it 

309 # only exists locally once a tutorial's video files have been downloaded 

310 # via --aprender. A fresh checkout (or a Docker build context, which 

311 # never sends it in) has no such directory yet - nothing to hard-link. 

312 videos_src_dir = f"{input_src_folder}/aprender/videos" 

313 for d in os.listdir(videos_src_dir) if os.path.isdir(videos_src_dir) else []: 

314 if d not in ["TODO", "blabla-bla"]: 

315 # Create a folder for that tutorial's video files 

316 # TODO: copy folder without content in order to keep the original folder's 

317 # creation date, in order to not confuse the rsync upload process 

318 os.mkdir(f"{output_prod_folder}/aprender/videos/{d}") 

319 for f in os.listdir(f"{input_src_folder}/aprender/videos/{d}"): 

320 # Create a hard link to the video file 

321 os.link( 

322 f"{input_src_folder}/aprender/videos/{d}/{f}", 

323 f"{output_prod_folder}/aprender/videos/{d}/{f}", 

324 ) 

325 

326 # Update links to leaflet (CDN) 

327 # Read the prod files 

328 with open(f"{output_prod_folder}/index.html", "r") as file: 

329 index_data = file.read() 

330 with open(f"{output_prod_folder}/404.html", "r") as file: 

331 page404_data = file.read() 

332 with open(f"{output_prod_folder}/aprender/index.html", "r") as file: 

333 index_aprender_data = file.read() 

334 # Replace the target strings 

335 # Leaflet 

336 index_data = index_data.replace( 

337 f'<link rel="stylesheet" href="leaflet/{LEAFLET_VERSION}/leaflet.css">', 

338 f'<link rel="stylesheet" href="https://unpkg.com/leaflet@{LEAFLET_VERSION}/dist/leaflet.css"\n integrity="sha512-Zcn6bjR/8RZbLEpLIeOwNtzREBAJnUKESxces60Mpoj+2okopSAcSUIUOseddDm0cxnGQzxIR7vJgsLZbdLE3w=="\n crossorigin=""/>', 

339 ) 

340 index_data = index_data.replace( 

341 f'<script type = "text/javascript" src="leaflet/{LEAFLET_VERSION}/leaflet.js"></script>', 

342 f'<script src="https://unpkg.com/leaflet@{LEAFLET_VERSION}/dist/leaflet.js"\n integrity="sha512-BwHfrr4c9kmRkLw6iXFdzcdWV/PGkVgiIyIWLLlTSXzWQzxuSg4DiQUCpauz/EWjgk5TYQqX/kvn9pG1NpYfqg=="\n crossorigin="">\n </script>', 

343 ) 

344 # Bootstrap 

345 index_data = index_data.replace( 

346 f'<link rel="stylesheet" href="bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">', 

347 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">', 

348 ) 

349 index_data = index_data.replace( 

350 f'<script src="bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>', 

351 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>', 

352 ) 

353 page404_data = page404_data.replace( 

354 f'<link rel="stylesheet" href="bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">', 

355 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">', 

356 ) 

357 page404_data = page404_data.replace( 

358 f'<script src="bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>', 

359 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>', 

360 ) 

361 index_aprender_data = index_aprender_data.replace( 

362 f'<link rel="stylesheet" href="../bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">', 

363 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">', 

364 ) 

365 index_aprender_data = index_aprender_data.replace( 

366 f'<script src="../bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>', 

367 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>', 

368 ) 

369 # jQuery (for Bootstrap) 

370 index_data = index_data.replace( 

371 f'<script src="jquery-{JQUERY_VERSION}.slim.min.js"></script>', 

372 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>', 

373 ) 

374 page404_data = page404_data.replace( 

375 f'<script src="jquery-{JQUERY_VERSION}.slim.min.js"></script>', 

376 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>', 

377 ) 

378 index_aprender_data = index_aprender_data.replace( 

379 f'<script src="../jquery-{JQUERY_VERSION}.slim.min.js"></script>', 

380 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>', 

381 ) 

382 # Bootstrap-toggle 

383 index_aprender_data = index_aprender_data.replace( 

384 f'<link rel="stylesheet" href="../bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}/css/bootstrap4-toggle.min.css">', 

385 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/gh/gitbrent/bootstrap4-toggle@{BOOTSTRAP_TOGGLE_VERSION}/css/bootstrap4-toggle.min.css"\n integrity="sha384-yakM86Cz9KJ6CeFVbopALOEQGGvyBFdmA4oHMiYuHcd9L59pLkCEFSlr6M9m434E"\n crossorigin="anonymous">', 

386 ) 

387 index_aprender_data = index_aprender_data.replace( 

388 f'<script src="../bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}/js/bootstrap4-toggle.min.js"></script>', 

389 f'<script src="https://cdn.jsdelivr.net/gh/gitbrent/bootstrap4-toggle@{BOOTSTRAP_TOGGLE_VERSION}/js/bootstrap4-toggle.min.js"\n integrity="sha384-Q9RsZ4GMzjlu4FFkJw4No9Hvvm958HqHmXI9nqo5Np2dA/uOVBvKVxAvlBQrDhk4"\n crossorigin="anonymous"></script>', 

390 ) 

391 # Copyright year in the pages' footer 

392 a = '<span class="text-muted">&copy; YEAR El Vallenatero Francés</span>' 

393 b = f'<span class="text-muted">&copy; {datetime.date.today().year} El Vallenatero Francés</span>' 

394 index_data = index_data.replace(a, b) 

395 page404_data = page404_data.replace(a, b) 

396 index_aprender_data = index_aprender_data.replace(a, b) 

397 

398 # Save edited prod files 

399 with open(f"{output_prod_folder}/index.html", "w") as file: 

400 file.write(index_data) 

401 with open(f"{output_prod_folder}/404.html", "w") as file: 

402 file.write(page404_data) 

403 with open(f"{output_prod_folder}/aprender/index.html", "w") as file: 

404 file.write(index_aprender_data) 

405 

406 # Create full HTML pages for Prod /aprender tutorials 

407 with open("../website/src/aprender/tutoriales.js") as in_file: 

408 # Remove the JS bits to keep only the JSON content 

409 tutoriales_json_content = in_file.read()[17:-2] 

410 tutoriales = json.loads(tutoriales_json_content) 

411 num_html_pages_created += len(tutoriales) 

412 for t in tutoriales: 

413 output_prod_tutorial_file = "{}/aprender/{}.html".format( 

414 output_prod_folder, t["slug"] 

415 ) 

416 shutil.copy( 

417 f"{output_prod_folder}/aprender/index.html", output_prod_tutorial_file 

418 ) 

419 with open(output_prod_tutorial_file, "r") as file: 

420 prod_tutorial_file_data = file.read() 

421 if t["author"]: 

422 tuto_title = "{} - {}".format(t["title"], t["author"]) 

423 else: 

424 tuto_title = t["title"] 

425 prod_tutorial_file_data = prod_tutorial_file_data.replace( 

426 "<title>Aprender a tocar el Acordeón Vallenato - El Vallenatero Francés</title>", 

427 f"<title>{tuto_title} - Aprender a tocar el Acordeón Vallenato</title>", 

428 ) 

429 prod_tutorial_file_data = prod_tutorial_file_data.replace( 

430 '<h1 id="tutorialFullTitle">TITLE</h1>', 

431 f'<h1 id="tutorialFullTitle">{tuto_title}</h1>', 

432 ) 

433 with open(output_prod_tutorial_file, "w") as file: 

434 file.write(prod_tutorial_file_data) 

435 

436 # Create full HTML pages for Prod / videos 

437 with open("../website/src/data.js") as in_file: 

438 # Remove the JS bits to keep only the JSON content 

439 videos_json_content = in_file.read()[16:-1] 

440 locations = json.loads(videos_json_content) 

441 num_html_pages_created += len(locations) 

442 for l in locations: 

443 # One page for each location 

444 output_prod_video_file = "{}/{}.html".format( 

445 output_prod_folder, locations[l]["slug"] 

446 ) 

447 shutil.copy(f"{output_prod_folder}/index.html", output_prod_video_file) 

448 with open(output_prod_video_file, "r") as file: 

449 prod_video_file_data = file.read() 

450 tuto_title = l 

451 prod_video_file_data = prod_video_file_data.replace( 

452 "<title>El Vallenatero Francés</title>", 

453 f"<title>{tuto_title} - El Vallenatero Francés</title>", 

454 ) 

455 prod_video_file_data = prod_video_file_data.replace( 

456 '<h2 id="list_location"></h2>', f'<h2 id="list_location">{tuto_title}</h2>' 

457 ) 

458 with open(output_prod_video_file, "w") as file: 

459 file.write(prod_video_file_data) 

460 

461 num_html_pages_created += len(locations[l]["videos"]) 

462 for v in locations[l]["videos"]: 

463 # One page for each video at that location 

464 # Create folder 

465 output_folder = "{}/{}".format(output_prod_folder, v["slug"]) 

466 if not os.path.isdir(output_folder): 

467 os.mkdir(output_folder) 

468 output_prod_video_file = "{}/{}.html".format(output_folder, v["id"]) 

469 shutil.copy(f"{output_prod_folder}/index.html", output_prod_video_file) 

470 with open(output_prod_video_file, "r") as file: 

471 prod_video_file_data = file.read() 

472 tuto_title = v["title"] 

473 prod_video_file_data = prod_video_file_data.replace( 

474 "<title>El Vallenatero Francés</title>", 

475 f"<title>{tuto_title} - El Vallenatero Francés</title>", 

476 ) 

477 prod_video_file_data = prod_video_file_data.replace( 

478 '<h2 id="list_location"></h2>', 

479 f'<h2 id="list_location">{tuto_title}</h2>', 

480 ) 

481 with open(output_prod_video_file, "w") as file: 

482 file.write(prod_video_file_data) 

483 

484 logger.debug(f"Number of production HTML files created: {num_html_pages_created}") 

485 

486 

487def generate_sitemap(sitemap_file, locations, uploaded_videos): 

488 logger.debug("Generate the Sitemap") 

489 base_url = "https://vallenato.fr" 

490 sitemap = generator.Sitemap() 

491 

492 # vallenato.fr index 

493 sitemap.add( 

494 base_url, 

495 # Timestamp of the most recently uploaded video 

496 lastmod=uploaded_videos[0]["publishedAt"][:10], 

497 changefreq="monthly", 

498 priority="1.0", 

499 ) 

500 

501 # Locations and individual videos 

502 sitemap.add( 

503 f"{base_url}/mundo-entero", 

504 # Timestamp of the most recently uploaded video 

505 lastmod=uploaded_videos[0]["publishedAt"][:10], 

506 changefreq="monthly", 

507 priority="0.8", 

508 ) 

509 for l in locations: 

510 # Locations 

511 sitemap.add( 

512 "{}/{}".format(base_url, locations[l]["slug"]), 

513 # Timestamp of the most recently uploaded video at that location 

514 lastmod=locations[l]["videos"][0]["publishedAt"][:10], 

515 changefreq="yearly", 

516 priority="0.6", 

517 ) 

518 for v in locations[l]["videos"]: 

519 # Individual videos 

520 sitemap.add( 

521 "{}/{}/{}".format(base_url, v["slug"], v["id"]), 

522 # Timestamp of that video 

523 lastmod=v["publishedAt"][:10], 

524 changefreq="yearly", 

525 priority="0.5", 

526 ) 

527 

528 # Aprender index 

529 sitemap.add(f"{base_url}/aprender/", changefreq="monthly", priority="0.9") 

530 

531 # Aprender: individual tutorials 

532 with open("../website/src/aprender/tutoriales.js") as in_file: 

533 # Remove the JS bits to keep only the JSON content 

534 tutoriales_json_content = in_file.read()[17:-2] 

535 tutoriales = json.loads(tutoriales_json_content) 

536 for t in tutoriales: 

537 tuto_url = "{}/aprender/{}".format(base_url, t["slug"]) 

538 sitemap.add(tuto_url, changefreq="yearly", priority="0.7") 

539 

540 sitemap_xml = sitemap.generate() 

541 

542 # Prettify the XML "by hand" 

543 sitemap_xml = sitemap_xml.replace("<url>", " <url>") 

544 sitemap_xml = sitemap_xml.replace("</url>", " </url>") 

545 sitemap_xml = sitemap_xml.replace("<loc>", " <loc>") 

546 sitemap_xml = sitemap_xml.replace("<lastmod>", " <lastmod>") 

547 sitemap_xml = sitemap_xml.replace("<changefreq>", " <changefreq>") 

548 sitemap_xml = sitemap_xml.replace("<priority>", " <priority>") 

549 

550 with open(sitemap_file, "w") as file: 

551 file.write(sitemap_xml) 

552 

553 

554def website(args): 

555 if getattr(args, "no_fetch", False): 

556 # Rebuild from the already-committed data instead of hitting the 

557 # YouTube API - locations/videos already have their slugs set, so 

558 # there's nothing left to do but regenerate the website files. 

559 locations = load_locations_from_website_data_file(WEBSITE_DATA_FILE) 

560 uploaded_videos = flatten_videos_from_locations(locations) 

561 logger.info( 

562 f"There are {len(uploaded_videos)} videos loaded from {WEBSITE_DATA_FILE} (--no-fetch)." 

563 ) 

564 else: 

565 # Retrieve the list of uploaded videos 

566 uploaded_videos = get_uploaded_videos(args, UPLOADED_VIDEOS_DUMP_FILE) 

567 logger.info(f"There are {len(uploaded_videos)} uploaded videos.") 

568 

569 # Identify each video's location 

570 (uploaded_videos, locations) = identify_locations_names( 

571 uploaded_videos, LOCATION_SPECIAL_CASES_FILE, UPLOADED_VIDEOS_DUMP_FILE 

572 ) 

573 

574 # Determine the geolocation of each location 

575 locations = determine_geolocation(locations, GEOLOCATIONS_FILE) 

576 

577 # Create a slug for each location (to be used for the website URLs) 

578 locations = determine_locations_slug(locations) 

579 

580 # Add the videos in each location array 

581 locations = add_videos_to_locations_array(uploaded_videos, locations) 

582 

583 # Generate the JavaScript data file to be used by the website 

584 save_website_data(locations, WEBSITE_DATA_FILE) 

585 

586 # Generate the development and production website files 

587 generate_website(locations, uploaded_videos) 

588 

589 # Generate the Sitemap 

590 generate_sitemap(SITEMAP_FILE, locations, uploaded_videos)