From f4d36e69534d9abc1dc7966a4b1cf8f7e7b10daf Mon Sep 17 00:00:00 2001 From: Junwei Zhao Date: Sun, 22 Mar 2026 16:15:53 +1100 Subject: [PATCH] Add image imports --- core/apps.py | 11 ++- data/db.sqlite3 | Bin 479232 -> 479232 bytes links/api_urls.py | 1 + links/api_views.py | 30 +++++++- links/file_views.py | 13 +++- links/serializers.py | 15 +++- links/tasks.py | 41 +++++++++++ links/templates/links/jobs.html | 97 +++++++++++++++++++++++-- links/views.py | 56 ++++++++++++++- static/llms.txt | 4 ++ static/openapi.yaml | 121 ++++++++++++++++++++++++++++++++ 11 files changed, 377 insertions(+), 12 deletions(-) diff --git a/core/apps.py b/core/apps.py index eceb15f..0125765 100644 --- a/core/apps.py +++ b/core/apps.py @@ -16,7 +16,7 @@ class CoreConfig(AppConfig): Initialize APScheduler when Django starts """ from core.scheduler import scheduler, start_scheduler - from links.tasks import schedule_pending_pages, schedule_pending_screenshots + from links.tasks import schedule_pending_pages, schedule_pending_screenshots, retry_stuck_image_imports from apscheduler.triggers.interval import IntervalTrigger # Start the scheduler @@ -49,3 +49,12 @@ class CoreConfig(AppConfig): replace_existing=True, ) logger.info(f"Scheduled periodic task: schedule_pending_screenshots (every {screenshots_interval}s)") + + # Add periodic job for retrying stuck image imports (every 5 minutes) + scheduler.add_job( + retry_stuck_image_imports, + trigger=IntervalTrigger(seconds=300), + id='retry_stuck_image_imports', + replace_existing=True, + ) + logger.info("Scheduled periodic task: retry_stuck_image_imports (every 300s)") diff --git a/data/db.sqlite3 b/data/db.sqlite3 index 1df67b1cef2c51dfec07779af731e36b949943d4..093307bb20ae5281d8e4942665a830f5db394cfd 100644 GIT binary patch delta 3180 zcmb_eU2GiH6`t9>v$M0Cotqz;K!}r=(Bc%vdw2fF5TTWv#8TPdJft>FkaK6|?vL3Y zXLpRFzAR&*Ril)5Osal~BwGk+Y4G1LE2P&aT2nivmq5>)pksK$qLPA0zaA!A8 zc5Meom9@I(YIW|p_ni68`ObOfbl)?l`(D_(F*d+3%rspO(0{l&Ea8%~E zKd2~U$ME<}HY@)oS9~;It5>kMXL73Wiz=>=Jr+X8=}PVcKI|AYz4FuE@v$ee*}X-# zzRaME>jjeVBVd+ii=H>t7)qxdQYLl<=Sz6dsa4W5Q<77wdZg;5r>Dv_>^9OelGSuZ zPLpz(NO@eXmZ}A*Q7lzSL&Bcq66}`gA8D#qZ+dlH@}!10?UrhxvQYDixFX>hTz4B% zc$1VFqyY$rfp6XPp-{)aE#V3-(5*^V^J8!5$G#hWjFt)>E|nWn!@*@T;D(pSL6l+n zP{cyPg#v^k8j46X$_Q_cM)^@T-in`&UVD5aFD4lH4m<%5i2oH|5f6%q^I7ywYX9D0 zj@j~fmPYqf10hwNz^*&ds7==$GBI5*Pk6P7e5p)2ABqO2OGuGqc_&haRDCF;4Qi$? zYgX&v_2_F~_Qj%15*W<7qWEg+-q=_0Nb-yT_|w8!U~t>$gZF@@Zzsf@@~i$R&7ycV z>%+X`$c~yl}wOBD))H$hYlWr&=&bW7J=*Pt8e zsBwX$C#Ogua_#PJpS7XP&$S^14a&&S6|_WkQ)6iD)Vz~S^W9`mG@Dec$(AM~TO*e0 zW?VaCE3)HeOhva96_b(So`mm;#@3Lf<{drb*a$n8tc)1>b|RH%)^x&QhaX$f=ZtL8 zUl%wu&`vZ|qZyrOdfr`DH#Z6`eunvyqO&8u=IHEO22Wu{kM9H-r6(hn#nR>g-vX)J z*jRrw&M<5?%LK~%!?m8J+weqTEIQbfI z<`ZNYj%B(?b~6+^N=k~c1ii>tXOqhyQp`b9H!VfEoklYj!Mj~2!v9w*y3G;5Q5SnP387?bnSxq;s>xZxc3S;z(#|~D?oz> zqN$uj4+u`Dq8?lT&#|fC!U9kMZTjy9UoU{S*83>Zd(YplfND^>0yYQeD%i2!@8?pv z2WVkZifO-7xe6v|GW_!@Ft;YfNk)7_{7jsLyWj}C2wSj1M~B~u&(p!-F|m1nG{>$+ zvB~hzYF#A}9$GEch44@}HQcolH?+RL4glN7{hmoO?=$gp@eS+#$h^;g$e#k=MZL%u z$xX4ZU)s){Nj|&T-^_23B=+P)YmVa<*liE(yzf@8y>I8!xb75jeLzO4xx4kR2={gb z2>aP1tv_(wIJ?)s$M5I29FkanPpRsX%Vj23+4MDE&ttEaDI z6x29wYqHC;)^j{}e0*yX&UMDRzr%Oo1$Z8wgRjEBz&ZF2ijEIqIC~^RkqCv*I#=U< zeUw^#(6L$w2SGX#UV@*&kKu>#18VhGxDiB!&`Joc1XiHpD$f2-c0SufhF(UuEN*8z?)sOX6VmhuUIE!nL|@{W6+x^RSNByvqIbw xPA?g$R&Jo}I~%vZPwu(!j;4#?Pcnc120A=ywBrpSZ4l{tM$uZI$owz+egsL{!Jz;E delta 832 zcmaiyO>7Kd7{}jt=AE6{&g?ubTTR4vx2;5qY2TUI`PhoYMril2+Qvs+(B17;A}9$- z4-Tf9_CkW-jSJ<_;LxC{so=oHMItUj(x{IENv+ZZiH|Z{4sH(r%kz7lJpbpP=jcf6 z=t%73(nZ1?Lg;XyHGwJu&74|oaKhY;?u*TM{Qw=Jc!+YNl&@SA2`H*W2tK1d@EyLu zr}3)p*P!IXID{e?W3IpdRtGP3;IJ=^@ITrsLux02yD$Ws)KBWD+8Ok(RsGI+{`#vL zXh)(fpvZgmLS&hehn@0OS*Xa3r)j2Qmsck#+2*ZrUP36GPUnr3+hgQx+brnGoSRJ9 zNz-tW`Bbl4aC%Y>r8;$7OWUL6HNEM;{@tckH0*-w6kXeL^L#!EL%^xmO%|$qo!u)37Tp#{}O{z!>bb&7M>v&D=p6sDcP>Ab!AghI?X0$ z$TjdB5)``@ih)yh@E1w2=v&go_P!-a@Upy;)ql3zh2mHk!UV=C9 z0_NZyyeppsJmIC`scFHVnhrH~h+GD7&$tXzsCXiZkhie`P@k5Gq|*^pC!utN9Z~uH zYy$X9^w^I2D%9mw4>tUR+ct#a?995*P03RrjH converter only captures the path component; query params + # like ?format=jpg&name=900x900 end up in QUERY_STRING and must be re-attached. + query_string = request.META.get('QUERY_STRING', '') source_url = f'https://{image_url}' + if query_string: + source_url = f'{source_url}?{query_string}' url_hash = hashlib.sha256(source_url.encode()).hexdigest()[:20] + # Strip query string for filename derivation filename = posixpath.basename(image_url.split('?')[0]) or f'image_{url_hash}' # Derive extension from filename; fall back to .jpg for bare names diff --git a/links/serializers.py b/links/serializers.py index b41ded0..b41098c 100644 --- a/links/serializers.py +++ b/links/serializers.py @@ -1,5 +1,18 @@ from rest_framework import serializers -from .models import Page, Post, ImageCollection, Image, Tag, FileUpload +from .models import Link, Page, Post, ImageCollection, Image, Tag, FileUpload + + +class LinkSerializer(serializers.ModelSerializer): + tags = serializers.SerializerMethodField() + + class Meta: + model = Link + fields = ['id', 'alias', 'original_url', 'description', 'link_type', + 'click_count', 'tags', 'created_at', 'updated_at'] + read_only_fields = ['id', 'click_count', 'created_at', 'updated_at'] + + def get_tags(self, obj): + return [{'id': tag.id, 'name': tag.name, 'slug': tag.slug} for tag in obj.tags.all()] class PageSerializer(serializers.ModelSerializer): class Meta: diff --git a/links/tasks.py b/links/tasks.py index de934d3..26e24d2 100644 --- a/links/tasks.py +++ b/links/tasks.py @@ -500,3 +500,44 @@ def download_and_save_image(file_upload_id): os.remove(dest_path) except OSError: pass + # Remove the stub record so a size=0 ghost doesn't appear in the file list. + # The next import request for the same URL will create a fresh stub and retry. + try: + FileUpload.objects.filter(pk=file_upload_id, size=0).delete() + logger.info(f"download_and_save_image: deleted stub FileUpload {file_upload_id} after failed download") + except Exception as del_exc: + logger.error(f"download_and_save_image: could not delete stub {file_upload_id}: {del_exc}") + + +def retry_stuck_image_imports(): + """Periodic task: retry imported images that are stuck with size=0 and no file on disk. + + A stub FileUpload (size=0, source_url set) can be left behind when the background + download thread is killed mid-flight (e.g. server restart). This task finds those + orphaned stubs and re-kicks the download, provided the stub is old enough that we + are confident it is not a currently in-progress download (>5 minutes since last update). + """ + from .models import FileUpload + + threshold = timezone.now() - timedelta(minutes=5) + stuck = FileUpload.objects.filter( + size=0, + source_url__isnull=False, + updated_at__lt=threshold, + ).exclude(source_url='') + + count = stuck.count() + if count: + logger.info(f"retry_stuck_image_imports: found {count} stuck import stub(s) — retrying") + + for record in stuck: + if os.path.exists(record.file_path): + # File landed on disk but DB wasn't updated — fix it now + size = os.path.getsize(record.file_path) + FileUpload.objects.filter(pk=record.pk, size=0).update(size=size) + logger.info(f"retry_stuck_image_imports: fixed size for FileUpload {record.pk} ({size} bytes)") + continue + + logger.info(f"retry_stuck_image_imports: re-queuing download for FileUpload {record.pk} ({record.source_url})") + thread = Thread(target=download_and_save_image, args=(str(record.pk),), daemon=True) + thread.start() diff --git a/links/templates/links/jobs.html b/links/templates/links/jobs.html index 6285383..e52dd06 100644 --- a/links/templates/links/jobs.html +++ b/links/templates/links/jobs.html @@ -147,10 +147,19 @@ {{ scheduled_jobs|length }} + + {% trans "Image Imports" %} + + {{ stats.image_imports.total }} + + - {% if tab != 'scheduler' %} + {% if tab == 'screenshots' or tab == 'pages' %}
{% for s, label, pill_active_style, pill_inactive_style in filter_options %} {% endfor %}
+ {% elif tab == 'image_imports' %} + {% endif %} @@ -176,7 +194,7 @@ style="background:#1d4ed8;color:#fff;padding:.3rem .85rem;border-radius:.25rem;font-size:.8rem;cursor:pointer;font-weight:500;"> ↺ {% trans "Retry" %} - @@ -367,6 +385,71 @@ {% else %}
{% trans "No pages match this filter." %}
{% endif %} + + {% elif tab == 'image_imports' %} + + {% if page_obj.object_list %} +
+ + + + + + + + + + + + + + {% for imp in page_obj.object_list %} + + + + + + + + + + {% endfor %} + +
+ + ID{% trans "Filename" %}{% trans "Source URL" %}{% trans "Status" %}{% trans "Size" %}{% trans "Updated" %}
+ + {{ imp.id|truncatechars:12 }} + + {{ imp.name|truncatechars:40 }} + + + + {{ imp.source_url|truncatechars:50 }} + + + {% if imp.size > 0 %} + + {% trans "done" %} + + {% else %} + + {% trans "pending" %} + + {% endif %} + + {% if imp.size > 0 %}{{ imp.formatted_size }}{% else %}{% endif %} + {{ imp.updated_at|timesince }} {% trans "ago" %}
+
+ {% else %} +
{% trans "No image imports match this filter." %}
+ {% endif %} {% endif %} @@ -461,9 +544,13 @@ function jobsManager() { if (this.selectedIds.length === 0) return; const tab = this.tab; const actionMap = { - retry: tab === 'screenshots' ? 'bulk_retry_screenshots' : 'bulk_retry_pages', - fail: tab === 'screenshots' ? 'bulk_fail_screenshots' : 'bulk_fail_pages', - delete: tab === 'screenshots' ? 'bulk_delete_screenshots': 'bulk_delete_pages', + retry: tab === 'screenshots' ? 'bulk_retry_screenshots' + : tab === 'image_imports' ? 'bulk_retry_image_imports' + : 'bulk_retry_pages', + fail: tab === 'screenshots' ? 'bulk_fail_screenshots' : 'bulk_fail_pages', + delete: tab === 'screenshots' ? 'bulk_delete_screenshots' + : tab === 'image_imports' ? 'bulk_delete_image_imports' + : 'bulk_delete_pages', }; if (type === 'delete' && !confirm(`Delete ${this.selectedIds.length} item(s)?`)) return; document.getElementById('bulk-action-input').value = actionMap[type]; diff --git a/links/views.py b/links/views.py index 1ff1280..1a5fe0c 100644 --- a/links/views.py +++ b/links/views.py @@ -586,7 +586,8 @@ class JobsView(View): PAGE_SIZE = 50 def _get_stats(self): - from .models import Screenshot, Page + from .models import Screenshot, Page, FileUpload + imports_qs = FileUpload.objects.filter(source_url__isnull=False).exclude(source_url='') return { 'screenshots': { 'total': Screenshot.objects.count(), @@ -602,10 +603,15 @@ class JobsView(View): 'completed': Page.objects.filter(process_status=Page.ProcessStatus.COMPLETED).count(), 'failed': Page.objects.filter(process_status=Page.ProcessStatus.FAILED).count(), }, + 'image_imports': { + 'total': imports_qs.count(), + 'pending': imports_qs.filter(size=0).count(), + 'completed': imports_qs.filter(size__gt=0).count(), + }, } def get(self, request): - from .models import Screenshot, Page, SiteSettings as SS + from .models import Screenshot, Page, FileUpload, SiteSettings as SS from core.scheduler import scheduler from django.core.paginator import Paginator @@ -623,6 +629,15 @@ class JobsView(View): if status_filter != 'all': pg_qs = pg_qs.filter(process_status=status_filter) + # Build image imports queryset + import_qs = FileUpload.objects.filter( + source_url__isnull=False, + ).exclude(source_url='').order_by('-updated_at') + if status_filter == 'pending': + import_qs = import_qs.filter(size=0) + elif status_filter == 'completed': + import_qs = import_qs.filter(size__gt=0) + # Paginate the active tab's queryset if tab == 'screenshots': paginator = Paginator(ss_qs, self.PAGE_SIZE) @@ -632,6 +647,10 @@ class JobsView(View): paginator = Paginator(pg_qs, self.PAGE_SIZE) page_obj = paginator.get_page(page_num) all_ids = [str(obj.id) for obj in page_obj.object_list] + elif tab == 'image_imports': + paginator = Paginator(import_qs, self.PAGE_SIZE) + page_obj = paginator.get_page(page_num) + all_ids = [str(obj.id) for obj in page_obj.object_list] else: page_obj = None all_ids = [] @@ -655,6 +674,11 @@ class JobsView(View): ('completed', 'Completed', 'background:#16a34a;color:#fff;', 'background:#f0fdf4;color:#166534;'), ('failed', 'Failed', 'background:#dc2626;color:#fff;', 'background:#fff1f2;color:#991b1b;'), ] + import_filter_options = [ + ('all', 'All', 'background:#1f2937;color:#fff;', 'background:#f3f4f6;color:#374151;'), + ('pending', 'Pending', 'background:#d97706;color:#fff;', 'background:#fef3c7;color:#92400e;'), + ('completed', 'Done', 'background:#16a34a;color:#fff;', 'background:#f0fdf4;color:#166534;'), + ] return render(request, self.template_name, { 'stats': self._get_stats(), @@ -666,6 +690,7 @@ class JobsView(View): 'scheduler_running': scheduler.running, 'site_settings': SS.get(), 'filter_options': filter_options, + 'import_filter_options': import_filter_options, }) def post(self, request): @@ -708,6 +733,33 @@ class JobsView(View): n = qs.delete()[0] messages.success(request, _(f'Deleted {n} page(s).')) + # ── Bulk image import actions ───────────────────────────────────── + elif action == 'bulk_retry_image_imports': + from .models import FileUpload + from .tasks import download_and_save_image + from threading import Thread as _Thread + qs = FileUpload.objects.filter(pk__in=ids) if ids else FileUpload.objects.none() + n = 0 + for record in qs: + _Thread(target=download_and_save_image, args=(str(record.pk),), daemon=True).start() + n += 1 + messages.success(request, _(f'Retrying {n} image import(s).')) + + elif action == 'bulk_delete_image_imports': + from .models import FileUpload + import os as _os + qs = FileUpload.objects.filter(pk__in=ids) if ids else FileUpload.objects.none() + n = 0 + for record in qs: + if _os.path.exists(record.file_path): + try: + _os.remove(record.file_path) + except OSError: + pass + record.delete() + n += 1 + messages.success(request, _(f'Deleted {n} image import record(s).')) + else: messages.error(request, _('Unknown action.')) diff --git a/static/llms.txt b/static/llms.txt index e344a46..b3e9cd1 100644 --- a/static/llms.txt +++ b/static/llms.txt @@ -4,6 +4,8 @@ ## What you can do with this API +- **List short links** (`/api/links`) — GET a paginated list of all short links with their aliases, target URLs, tags, and click counts +- **Most visited links** (`/api/links/most-visited`) — GET links sorted by visit count descending (only links with at least one click) - **Bookmark pages** (`/api/pages`) — POST a URL and the server auto-fetches the title, description and screenshot in the background - **Write blog posts** (`/api/posts`) — Create Markdown posts with tag categorisation - **Manage image collections** (`/api/collections`) — Group images into named albums; upload multiple images at once @@ -43,6 +45,8 @@ No authentication is currently required. | operationId | Method | Path | Description | |---|---|---|---| +| listLinks | GET | /api/links | List all short links | +| listMostVisitedLinks | GET | /api/links/most-visited | Links sorted by visit count | | listPages | GET | /api/pages | List bookmarked pages | | createPage | POST | /api/pages | Bookmark a new URL | | listPosts | GET | /api/posts | List blog posts | diff --git a/static/openapi.yaml b/static/openapi.yaml index f84d7d4..e267e35 100644 --- a/static/openapi.yaml +++ b/static/openapi.yaml @@ -50,6 +50,8 @@ servers: description: Local development server tags: + - name: Links + description: Short links with click tracking. Supports listing all links and retrieving the most visited ones sorted by click count. - name: Pages description: | Bookmarked web pages. Creating a page triggers an async background job that fetches @@ -69,6 +71,62 @@ tags: Files can be made public/private and given an expiry date. paths: + /api/links: + get: + operationId: listLinks + tags: [Links] + summary: List all links + description: Retrieve a paginated list of all short links ordered by creation date (newest first). + parameters: + - in: query + name: page + schema: + type: integer + default: 1 + description: Page number (1-based). + - in: query + name: page_size + schema: + type: integer + default: 10 + maximum: 100 + description: Number of results per page. + responses: + '200': + description: Paginated list of links. + content: + application/json: + schema: + $ref: '#/components/schemas/LinkList' + + /api/links/most-visited: + get: + operationId: listMostVisitedLinks + tags: [Links] + summary: Most visited links + description: Return links that have at least one click, sorted by click count descending. + parameters: + - in: query + name: page + schema: + type: integer + default: 1 + description: Page number (1-based). + - in: query + name: page_size + schema: + type: integer + default: 10 + maximum: 100 + description: Number of results per page. + responses: + '200': + description: Paginated list of links sorted by click_count descending. + content: + application/json: + schema: + $ref: '#/components/schemas/LinkList' + /api/pages: get: operationId: listPages @@ -969,3 +1027,66 @@ components: enum: [success] message: type: string + + TagSummary: + type: object + properties: + id: + type: integer + name: + type: string + slug: + type: string + + Link: + type: object + properties: + id: + type: integer + readOnly: true + alias: + type: string + description: Unique slug used as the short-link identifier (e.g. /gh → github.com). + original_url: + type: string + description: Target URL. May contain template parameters like `{param,default=value}`. + description: + type: string + nullable: true + link_type: + type: string + enum: [LINK, CUSTOM] + description: LINK redirects to original_url; CUSTOM renders Markdown content. + click_count: + type: integer + readOnly: true + description: Number of times the short link has been visited. + tags: + type: array + readOnly: true + items: + $ref: '#/components/schemas/TagSummary' + created_at: + type: string + format: date-time + readOnly: true + updated_at: + type: string + format: date-time + readOnly: true + + LinkList: + type: object + properties: + count: + type: integer + next: + type: string + nullable: true + previous: + type: string + nullable: true + results: + type: array + items: + $ref: '#/components/schemas/Link'