From cb3dfed6d035a190df21ea1e0652daf7f30bff88 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Mon, 27 Oct 2025 17:55:03 -0700 Subject: [PATCH 01/10] Add docx export --- lumen/ai/assets/lumen_template.docx | Bin 0 -> 10007 bytes lumen/ai/report.py | 231 ++++++++++++++++++++++++++-- lumen/ai/schemas.py | 7 +- 3 files changed, 225 insertions(+), 13 deletions(-) create mode 100644 lumen/ai/assets/lumen_template.docx diff --git a/lumen/ai/assets/lumen_template.docx b/lumen/ai/assets/lumen_template.docx new file mode 100644 index 0000000000000000000000000000000000000000..756a927c7f6acc8a5f5c732de1cda6d35443e923 GIT binary patch literal 10007 zcma)i1yEee*7e}-PH-9AC3tXmmjugTgS)%?;O=e-?hpv>?rwpg0TLh~fuH2Q_ik?L z|6aX4RZ}xlwN9VYy=V2>YqzQb3@k1H5fKsK6JcNg_$@G>KD*mHf!Hj~Oh9H%9IPI; zHWR56cB`V8LwWfZ#%H89MYtvS!0Jxwis8pTXW-I#R%WF%BL+ z?+RZY>Fl<*|0ekw(fbhyGK62X2Ep?jEP9<3y72b%9jo;~&(MD>b|7n*lnhcuD?N%v zQN{5-eXS!USh$K{Wo>dT5}EuaZiW4dwP9UHc{qMRmaF(-@1CJ$>z=@6VzFR#0fkb)4`pnD`#2zhn$ zpjt*hb~IcX;iQF$_YXrV9c&Q{9jy51u^%jg^z6l`YMw9EM*E5L{c4H#Q}^PMzMfc% zHaT@$s=zOPXV~lEgSWR2olgnioO{dWwHOtvj88J0GDUBYQtp2O_lQ!jkZ%&dTUTjga)rAc=gXZmO|dz zL}-UN^|1FnKJR0X(Rs?UYxy|oav|W35Cz#(g(;Pa>DBGj?TYYaOlY=PsS{Ze_49p` zA6Bjr>S4J!hrmQ`U`3O_m9K;bYwvi08_d>kX_a}?nsd2sEn9_T&U4=GG$ zxnMO82KIR3ensS{(ej)cnuOIe*n^2#MhC@p{d%=Vh)_#egvizSLB3NGvaNn*w!HIX zV`wsYmn1{OO7USz4}6v+d-jbs7wZ$>I5cqn`sLe!a@KANz7&>P>tP1OOz2g7 zh!Q0z>YV=`-V^B+R|7~>&_kY#qDAoSBh=62dW$R*ME;aq8wmeXa+%xPyZnp9 zYEswih&eDt-{n8nUU}D}$QQZbMO02pr+f>T-RI}LIo%z69*YJm_4RhuJ6JRdU!Uv~ z=N`ZF-0f{kciKV4ywp&O=rf~Ff6@b>gRhZOI4kp z(qUhpISojI&bub&=Ii+Bk0)c{NRvawq_U5}E8HON zQ=&VT1%*`}E;jF`YxGDIeJZUipsP&IDaD#qnvpZuM1tut)-BxJ$2bdBH1jj^d_3+$m2lBv*u5M!49x791XQPZ4=bX%n%dp7Nb{Jh<4~OLvX^m4#}U zoYooUs;5r?0r^Repn~4C0W3l*EN8A?*F8Swik0tsnhRvZ9 z`Up*fb9#8wMw~vrO6fBqpC^+_Xoc_wJ$m-fSg3rSSqWty3w~E3`YS{6jl#!4PZ|2* zf1|&@GV~Yyb*27;{^(&+%|kH8`brgha_wBxx1RIlet@vZqjpaehvEV}{_4f|&x}TN6=Uxpa4hyXBoT{x<#vWpCNaw|F&$_e41@EEzrf#9|VS0{5 zdipR$k50J5E|L#PQrCy{TuBxugB=0U6Pw8)Y<{zfy`%HK(0v6yNqc^&tJqD5>!O3j z#KV?*&G=`e)GgGsDjdj}Fg&F!2NHXBmR_r~xa6%}J1f1R951DbVIC417ajEI>VlmD zN$;|nwv7Cfhhsz7`gr=mb?Y1I+t|U!n6X2tB{4CB{b<@~^Q!dptDRk*l5I8L^!8#z z%H;KsH}Aa}8k@eq+t)#0L{ZP#*b101`pJO{L{lO(PZ=itUo#B*-$KsL)z;L^$;!^+ z*A|%lbrGin7GIf7Nw)MmWCnal5iAWI&+#v`iLKRK(|YjZt{!w zH_ZyXVYL{(+E>|@_7FYpWI+OLokL>@k|9!JT`FR!G;F^Q)l6y>F2M{Nj&NY7c=7W} zD>_YYNHc-58^b<{7lAdb(55z&@Hfwjmcpdyw}@f~C^o1o2x88nMlc}qAa0kj6_g`V zZgX=zIzgj^^Z-<`CZMbPj}=EBv$S{;W_zoZM@xH4o* z&C>mAn+tK6A}WFc05akK+ct;w?eoT`wx`g4ULf~u>*I75*Bpi5&2DWNjVW~;q%EmTn#&ZTrBK0)P{DIiLHae5 ziSD}m^IimJ)TtT|7gG@NVZj)-7#U{9QZ5RxaXG`9@RFG2*K6Ar4(CTXSmLW=?Zkx6 zljyfBnDrNk%!A`-bf#ysIl(Gw#6#B8uR-u&7G7b~Nru_k;rHJlvCPO(vX3}apOZUZ z4QEO6(7kUaqE<>548|=j9EBzJPyP_26+rU%1yg1+$f&3k)JM$wW;)w7lDML778~nz z_P0eM1XFAk`g|EwER8sa46f`D2}bxB`#7d)i1&`ut(3LeUT{nkyUnPWz#Lnqb-7AQ zSpdfTwUFVA+$$d`2esx!2?>!PQM`Fxz%n%~a3*m3LX~7^db~Azeo}`<(WHWsJZpaa z%M|cZtK*chd{~z0G5$tkOh5Ll+b6{B1|C>-5BP{?>`T7=6+dA3@O8|vRIW39cYFgG zANP2$6#!+T*W-N);;lw)N|?^J8~tO-UaH7q(d;rBC}K4bwfWqSQi#sut#$nihAY;b zrRo)uf@e5L5uGqVxay3JJ|Bw`lqz_<}c6M@4*4p0S3s!8|&?@aUwLpO-*Zo zu1pa#mxQE`IT%#rsuirZhCm|Y#%i10uZw539q4OHhNWFBwzgIl*3U61b*K+F*rZv| zRx`FVrgNoji5d^eYl51In_QIENS9u~F%m0mV%L5!3U+IM5f0+|T5{~XEZmsb48?}a(hUG7F%j`P zl+tXVh0W|_XH|@XiAj#+#}?{HUf|+t`VWBs5LE8mb4ix$Rm=sM|ga+7kXDVdNrX)Wihy+Q9QbP zOdowz=A5lKbiY9QWRmzS7ao^d8CLkjac;${oC^o3+$}RciNuNI1;aR|DC_p%bI_M2 zfAXK~#zvb(O*BaI3=&>honKr z%sGa$B>|nomoFkc-{5Q8Fg|&p81LGW1xp{$%3R&^Wd%~-`>~^8_FVWX+(~;b!Kp~} zVYS+hUO4SqSJ|qaaPxwU%fb}hHVzWGGoVNX$J~?i5jSkD+)qv)evwfA>ulGSClYG; zuOx)=9};r$v@!cdK^tk0cDsB}b)4}Q)WIV!rD&W)iB`+PW=`V^<(4i4LTm|UlC^qL z`uoq;ba6;X`4H_@w;hb#y2F!8IK`~1vYBx14#oQHmhz-@%-XY z{Tu^&FghSM0X)Kw6QZQWh(h5BRJ`!E9Kx4|N3pl9R4V`Qy5(JbN`P=$9dg=$a&zas*?}(W|!d&ly z-?Wlw;7f&~Pi1+_uWH%bur2LJWn4cOUGk`#RM5C>AER;p#`$;McM<$jsJm583 z84OI(m3ASl5e*aWwYjsgTh_1XFv_I=0edwwM!gla zs7?x++Usjuc0Vr3O&h#S64RHmm^NWm@Wh*tDS@+NDdRw8^l->_}ZZc z_pIYrwk28ALZ;Po$m@*d>rTHD2I_p&BYI?K--0_`J7~Fdz%;kt`v_I2>{1TnT=9Ji zaontuB_owIxF8f7n|8}`!N4B}SGel6Tb7n}$lcN^CX?lOo~!vtSWX|Z@BBdXL&Rr} z?e|DFs~xzZCF1RnaFbp?HPl@1Jqo_O`Zof%v^?mmv|6j+gE<7S2R8WbO@3VA8&v`& zGXzf;&=#XDTZoVVlIQ1UN)-&Ea9^UK81|hjH&Azt#A~np>GW>s@0W6Sd$+JJvfV=c z0fLMoEX#Y;wL@%S*-T_aOmI?^o$0@&$^HW)FO)Cp@y2_mjvo z^IWM@U;uyv-2X{dfb3tp+M3z9{N(X7eOvoAUd#u$dyJ3(+-KXmsmy|L24zQyr7Y7m z(n)?Os;by#b)}M_#npULkN2`_g$PHa{1-C!!>#1P}~xLWC~p#SJ}V&k@@j z-vP0?8%iK^WV5F;XD7cHtyi{~c=g6Ky#Q`?f2*MDHTM+$`8jD&vkP?2<{n@jE#P4f z0G|bbY8OQ9OHCZ(cb*{E$~-OCR}}Rt{F|rWd;+P`aq7%KQXHzf;20I;CGb7g9@bnBSrZ#6BX$WY!Z=^sD^(Qgt+xlae(3m&ABPub4 zMxy2#Q4e*6etwn%|6XHbG?Ns~sIN<@ZvpBC6UTtiV8gz|Qx8$yzAB`7Le{ev!xJ8+ z0yKyzV-r^0aIqDrdXMnJuY5#2+7g}EnAg2!<&Be(P|bm|+|Cf`7L%Q?uEFGr8Gjj6 z>K+~(8jqp-ipL53w~zfiQONeO$XcK7_3iERb0W9$t{1vXCcZOnk~#3fbMlE9`4a7v zorv5THCSHgNCOEsOA?}2Y}w8h5=J)gGS&~{6qlycK+}lXbDHIy#?}tI1?9+SVSQeB zOCP;9-Z;;G`K%1G<4k?&d-aZ`X;VxbGb>BM$9Dp6hK-?nBGxw>Ofc4ZlL)L=FW~N( z`ZngLy$!*ae=e7>($hW;)L>oaX1<8w+5wb7nN`mQnTX2PltOEx5KEWO}tM@VLG`$nbi*!1cwiYHQXK zJSH8FF#kbTPb@#PhE7=JKK&>_;}-Z-oxDc(d&f*3wml%(2|GKopurvurDGEH2$heD zPbGcz*;pJf&Pl`H!;n6g4W&k2k}n{CMu+iP&{D>#)%fKQlJ^(BRiK^Fb)AssC-IA$ z=V(>;5AUY4yb3UKc?>pei;G@ks=@Oyk}n)5Yips$5og@wolnM8-p%{AkCz|q&x|C) z8*1XZn$K^f5V__Fa;e+%+IG4nk=xl5d*R}r?nn-@jF3Uz0Xgq1G>EtZi%EN`QRydK zdauKFgs@GQ^l^^fB`>SDu4$sI)LI5oMC%b{@}Q3gEFfV8rh%gK zBm*XJ!T4oSqKz0Q45wJL5&GkhzvrY8FTp#wnx3dkmllZ_f?ysmlTWvHv0=^)Bh(?S zg#fv)iTzxU$NZ7Kh%8SX{P2`l3Em+SZRIl`p5!&%8hCX3cegxz`C;_f_~PBS5wjOR zC-}0E*@oarpfW#A^8ejwKzW+rzvlMWe8x@*qQ7THh2O?;2p<+y8VkS#&%#l7YSOknZBohGt}iJUp9@uHtcwBB`RwIk;Ny%geo*s@_8%u>9~yc=1b$vsg}1 z)jd52vGR)pwSMVup%sl_DD7 zRQuocnst_nV`IU7u6x;M%cRMu*I@gxmot^XQ{2>pcbDotaw%1mtn@s!C8WM_v25HMwcC_-4lX|O6|px-3(qCs$XJTqs7utu;gj?yJ0v&hW3A(I}>TFi!Jcu!*R((qQCIryNHu7 zUK_a@S7U(X>mGFrsI}9U?8)RZ_|fp_&JLcI#p+$2hgi7p;IcVK9j3cM)UB62EnudJ zSTDG(?41p%AA-*|&gJ1QCOmq?qI71&T}Wb-}q1PjPb6g8El@s5{|4EpkJd z(;JMv$rjNbb7Faa+j;~;fx@cIEe%eDVBG+g24c|Rl;E~`f|}^QCejh4s_V&m1pzB< zeY1_rBZ&478pw@Ayo2J-UkJB9j?Dxk$lUt#s=n%T%-86;dx(BV?i=;sb+r*4Zj^DU zKJ@r7XYuaGw-Taqm+kAfZY_ zrkJLc8aDlb{T#|xs7JZTOh;KbQ3%dol$)yL9z%Tj`P=%$-qnlQ+2glfkt%a<(6fdN~+L8 zZ{R-Kcw5~1zV`;EOr)9xak#(@bsGXh)C{kVG_6OI_$+@v)R01-Z#cGV^J=$J@N_oC z$D}&;6xf{wuk2bZpySLAQv) zN+9BU-*F8j8(M=!-t~3%+sfV=_|RFMJN1rss=c;`ypfA{W>(`ig0r~_7_$!1drURX z-3lMDNyJJC|4tKPLir{L1$J^!*5M11){b2S4(1*R8(OaxokqSjq$;#2+1>3$iLrC@ z2W7UDfI)Ed{Z}W6cQ&au^?DSpVzJ4+R-rtsUAgRZ2)XK+dzmF|L-Gwg_Tn29+1>IE z+Q)5{vbf_BdEbOB*wMzGp(bd|nc|w}5vKCoXAeHREcFJ`P7aMU!O^WnFZAyoqmnk( zXtkW{5BJHf5%*dlg{eV*2d!GMqwF>xRIK+TSvyYdYB_83^SEd9;$cNXz&K;u)wr+gMoit6?J z`t+(IMA;}$8Ft^TLADdTgWtsQI9ggxG@gh_u(CO8n%CFhSRuhXH{XzU^bl3bJvoDW zyTL;{@scl(_xtHAY)5Mk{lrZ3KqxZ5O94%>p;1qF-UxITX0lWzw&_QdBTb$97bgb% z@tq@Z1*-*6*th~Y$Z(Wvm9^Kf=E|?O$jV}(L^LXylD@mk-XJ|IUg$_%wJ&g3RYH~l zwSLWCDLp2J|Iu={J?m2vYXL0xOr3aJ!Q*Sor0Xk;r`!N#y?bmPeR4jVW8qh&w4E9_ z%VxQ*>0#q0MEZ#|LWJ?w6E@PLKr7j(jSn`*Ak&ed%Krp+ULX77T1XH)X~*n%bGd7{ zvGS&O>HP9dGl9bV;yLAt%}zYMNGC%5^BpfP^RH2g#;F+1Qs$o^_erM1piOKi8AStu zX!l`6wMPfyMgDmF$Ir=GUzJ%Yn!0FgvK4i=EwW_gUcSg^El%IOl;1tYw=nH7PjvPn zJE)gR%{9F|X+6aq&3a`%zB$)^iW}$!msZiY_e$Gil$Vx_UlZEB%>R|x!)inFtH)o; zg;KvTg!#$!StI>#+YQdsc4Hv%WcEFkh(?;84rb1;epZSDDGJJcKOMgwQYw43y=IgU z)j8Gx)<7x#0Be;@#CISSd2eqZ29+8Qv>Fv}LXN-rgT2j=UNyQjoi}2g3-9X&mr?B6 z*^1e7&YN~U6+1m_HC@`q21p&XWV5~@3$F|YM*-*=mo~*W!O0@#o``e-jMU_HIBH0# z$xs)>-W0({Bxlp%BXiII(jw2Y(_J*!I3INCTqiS!v83? zxmk&m9yxZc9^_QdnVi^eYbX^3-&9Ivq?j!}?r%N!B@2F9Ns(Ee{H@A81a1uH#yb_k)>MT{~v#ca8=FBVi@NI`vP+C6hPjof2Wzm#_9Nc3xyi7#0qZ z#|6>Ac0GRE`?1@9^Sfs1FEbX=fm7Y!Nd_eUa^;|)aRGlH io.BytesIO: + """ + Export the report to a Word document (.docx) format. + + Returns + ------- + BytesIO + A BytesIO buffer containing the rendered docx document. + + Raises + ------ + RuntimeError + If the report has not been executed yet. + FileNotFoundError + If the template file is not found. + """ + from docxtpl import DocxTemplate, R + + # Validate execution + if len(self) and not len(self.outputs): + raise RuntimeError( + "Report has not been executed, run report before exporting to_docx." + ) + + # Load template + template_path = Path(self.docx_template_path) + if not template_path.exists(): + raise FileNotFoundError(f"Template file not found: {template_path}") + + doc = DocxTemplate(str(template_path)) + + # Start with copy of docx_context + context = dict(self.docx_context) + + # Set defaults for missing keys + if 'title' not in context: + context['title'] = self.title or "Lumen Report" + + if 'subtitle' not in context: + date_string = datetime.now().strftime("%B %d, %Y") + context['subtitle'] = f"Generated on {date_string}" + + if 'cover_page_header' not in context: + context['cover_page_header'] = "" + + if 'cover_page_footer' not in context: + context['cover_page_footer'] = "" + + if 'content_page_header' not in context: + context['content_page_header'] = "" + + # Always generate sections from report outputs + context['sections'] = self._generate_sections(doc) + + # Always set page_break + context['page_break'] = R("\f") + + # Render template + doc.render(context) + + # Return as BytesIO + buffer = io.BytesIO() + doc.save(buffer) + buffer.seek(0) + return buffer + + def _generate_sections(self, doc) -> list[dict]: + """ + Generate sections list from report tasks for docx template. + + Arguments + --------- + doc : DocxTemplate + The document template instance (needed for InlineImage creation) + + Returns + ------- + list[dict] + List of section dictionaries with title, image, and caption + """ + from docx.shared import Mm + from docxtpl import InlineImage, RichText + + sections = [] + + breakpoint() + for section in self._tasks: + if not isinstance(section, Section): + continue + + section_dict = { + "title": section.title or "Untitled Section", + "image": None, + "caption": RichText("") + } + + # Process section outputs to find visualizations and captions + image_found = False + for i, out in enumerate(section.outputs): + if isinstance(out, LumenOutput) and not image_found: + # Convert LumenOutput to image + image_path = self._output_to_image(out) + if image_path: + section_dict["image"] = InlineImage(doc, image_path, width=Mm(160)) + image_found = True + + # Check if next output is a Typography for caption + if i + 1 < len(section.outputs): + next_out = section.outputs[i + 1] + if isinstance(next_out, Typography): + section_dict["caption"] = RichText(next_out.object) + break + + if section_dict["image"]: # Only add section if it has an image + sections.append(section_dict) + + return sections + + def _output_to_image(self, output: LumenOutput) -> str | None: + """ + Convert a LumenOutput to an image file path. + + Arguments + --------- + output : LumenOutput + The output to convert + + Returns + ------- + str | None + Path to temporary image file, or None if conversion failed + """ + try: + # Create a temporary file for the image + tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False) + tmp_path = tmp.name + tmp.close() + + # Render the component and save as image + component = output.component + + if isinstance(component, (View, Pipeline)): + # Render the view/pipeline to a panel + panel = output.render() + if hasattr(panel, 'save'): + panel.save(tmp_path) + return tmp_path + elif hasattr(component, 'save'): + component.save(tmp_path) + return tmp_path + else: + print("Output component is not renderable to image.") + + # TODO: add vega-lite export and table to docx table + breakpoint() + + # If we couldn't save, remove the temp file + import os + os.unlink(tmp_path) + return None + + except Exception as e: + print(f"Failed to convert output to image: {e}") + return None + class Action(Task): """ @@ -733,8 +939,8 @@ class SQLQuery(Action): and generates an LumenOutput to be rendered. """ - generate_caption = param.Boolean(default=True, doc=""" - Whether to generate a caption for the data.""") + schema = param.Dict(default=None, doc=""" + Optional schema to use to not infer schema from data.""") source = param.ClassSelector(class_=BaseSQLSource, doc=""" The Source to execute the SQL expression on.""") @@ -749,8 +955,12 @@ class SQLQuery(Action): table = param.String(doc=""" The name of the table generated from the SQL expression.""") - user_content = param.String(default="Generate a short caption for the data", doc=""" - Additional instructions to provide to the analyst agent, i.e. what to focus on.""") + template_overrides = param.Dict(default={}, doc=""" + Template overrides to provide to the AnalystAgent.""") + + user_content = param.String(default=None, doc=""" + Instructions to provide to the analyst agent, i.e. what to focus on; + if unset no additional instructions are provided.""") def _render_controls(self): return [ @@ -800,7 +1010,7 @@ async def _execute(self, **kwargs): # Pass table_params if provided params = {self.table: self.table_params} if self.table_params else None source = source.create_sql_expr_source({self.table: self.sql_expr}, params=params) - pipeline = Pipeline(source=source, table=self.table) + pipeline = Pipeline(source=source, table=self.table, schema=self.schema) if self.memory is not None: self.memory["source"] = source if "sources" not in self.memory: @@ -808,12 +1018,11 @@ async def _execute(self, **kwargs): self.memory["sources"].append(source) self.memory["pipeline"] = pipeline self.memory["data"] = await describe_data(pipeline.data) - self.memory["sql_metaset"] = await get_metaset([source], [self.table]) self.memory["table"] = self.table out = LumenOutput(component=pipeline) outputs = [Typography(f"### {self.title}", variant='h4', margin=(10, 10, 0, 10)), out] if self.title else [out] - if self.generate_caption: - caption = await AnalystAgent(llm=self.llm).respond( + if self.user_content: + caption = await AnalystAgent(llm=self.llm, template_overrides=self.template_overrides).respond( [{"role": "user", "content": self.user_content}] ) outputs.append(Typography(caption.object)) diff --git a/lumen/ai/schemas.py b/lumen/ai/schemas.py index 181638abb..30455b660 100644 --- a/lumen/ai/schemas.py +++ b/lumen/ai/schemas.py @@ -222,7 +222,7 @@ def __str__(self) -> str: return self.table_context -async def get_metaset(sources: list[Source], tables: list[str]) -> SQLMetaset: +async def get_metaset(sources: list[Source], tables: list[str], schema: dict | None = None) -> SQLMetaset: """ Get the metaset for the given sources and tables. @@ -232,6 +232,8 @@ async def get_metaset(sources: list[Source], tables: list[str]) -> SQLMetaset: The sources to get the metaset for. tables: list[str] The tables to get the metaset for. + schema: dict | None + Optional schema to use instead of fetching from sources. Returns ------- @@ -253,7 +255,8 @@ async def get_metaset(sources: list[Source], tables: list[str]) -> SQLMetaset: source_name = next(iter(sources)).name table_name = table_slug source = next((s for s in sources if s.name == source_name), None) - schema = await get_schema(source, table_name, include_count=True) + if schema is None: + schema = await get_schema(source, table_name, include_count=True) tables_info[table_slug] = SQLMetadata( table_slug=table_slug, schema=schema, From 247b9896617c61ddd8101c118e4fc88a08075776 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 15:10:48 -0700 Subject: [PATCH 02/10] Add multimodal support for llm --- lumen/ai/llm.py | 25 ++++++++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/lumen/ai/llm.py b/lumen/ai/llm.py index d684fd899..4fc01c0f4 100644 --- a/lumen/ai/llm.py +++ b/lumen/ai/llm.py @@ -13,6 +13,7 @@ from instructor import Mode, patch from instructor.dsl.partial import Partial +from instructor.processing.multimodal import Image from pydantic import BaseModel from .interceptor import Interceptor @@ -25,6 +26,13 @@ class Message(TypedDict): content: str name: str | None + +class ImageResponse(BaseModel): + # To easily analyze images, we need instructor patch activated, + # so we a pass-thru dummy string basemodel + output: str + + BASE_MODES = list(Mode) @@ -168,9 +176,22 @@ async def invoke( kwargs.update(input_kwargs) if response_model is not None: - if allow_partial: + if allow_partial and isinstance(response_model, BaseModel): response_model = Partial[response_model] kwargs["response_model"] = response_model + # check if any of the messages contain images + elif response_model is None and ImageResponse: + for message in messages: + if not isinstance(message["content"], list): + continue + for item in message["content"]: + if not isinstance(item, Image): + continue + kwargs["response_model"] = ImageResponse + # Currently instructor does not support streaming with multimodal + # https://github.com/567-labs/instructor/issues/1872 + kwargs["stream"] = False + break output = await self.run_client(model_spec, messages, **kwargs) if output is None or output == "": @@ -179,6 +200,8 @@ async def invoke( @classmethod def _get_delta(cls, chunk) -> str: + if isinstance(chunk, tuple): + return chunk[1] if chunk.choices: return chunk.choices[0].delta.content or "" return "" From 8f454daa9a87584120d188c7c6c3e540d1143932 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 15:19:06 -0700 Subject: [PATCH 03/10] Add multiomodal support --- lumen/ai/agents.py | 3 ++- lumen/ai/llm.py | 6 ++++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/lumen/ai/agents.py b/lumen/ai/agents.py index 9e5519eb8..4c3716659 100644 --- a/lumen/ai/agents.py +++ b/lumen/ai/agents.py @@ -130,7 +130,8 @@ def __panel__(self): async def _stream(self, messages: list[Message], system_prompt: str) -> Any: message = None model_spec = self.prompts["main"].get("llm_spec", self.llm_spec_key) - async for output_chunk in self.llm.stream(messages, system=system_prompt, model_spec=model_spec, field="output"): + output = self.llm.stream(messages, system=system_prompt, model_spec=model_spec, field="output") + async for output_chunk in output: if self.interface is None: if message is None: message = ChatMessage(output_chunk, user=self.user) diff --git a/lumen/ai/llm.py b/lumen/ai/llm.py index 4fc01c0f4..052cab4fb 100644 --- a/lumen/ai/llm.py +++ b/lumen/ai/llm.py @@ -200,8 +200,6 @@ async def invoke( @classmethod def _get_delta(cls, chunk) -> str: - if isinstance(chunk, tuple): - return chunk[1] if chunk.choices: return chunk.choices[0].delta.content or "" return "" @@ -284,6 +282,10 @@ async def stream( model_spec=model_spec, **kwargs, ) + if isinstance(chunks, BaseModel): + yield getattr(chunks, field) if field is not None else chunks + return + try: async for chunk in chunks: if response_model is None: From 0d2878ab1abb326fceaa94642d991889da751cc4 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 15:33:55 -0700 Subject: [PATCH 04/10] support single content --- lumen/ai/llm.py | 33 ++++++++++++++++++++------------- 1 file changed, 20 insertions(+), 13 deletions(-) diff --git a/lumen/ai/llm.py b/lumen/ai/llm.py index 052cab4fb..268290bbe 100644 --- a/lumen/ai/llm.py +++ b/lumen/ai/llm.py @@ -29,7 +29,7 @@ class Message(TypedDict): class ImageResponse(BaseModel): # To easily analyze images, we need instructor patch activated, - # so we a pass-thru dummy string basemodel + # so we use a pass-thru dummy string basemodel output: str @@ -130,6 +130,17 @@ def _add_system_message( messages = [{"role": "system", "content": system}] + messages return messages, input_kwargs + def _check_for_image(self, messages: list[Message]) -> bool: + for message in messages: + content = message.get("content") + if isinstance(content, Image): + return True + elif isinstance(content, list): + for item in content: + if isinstance(item, Image): + return True + return False + @classmethod def warmup(cls, model_kwargs: dict | None): """ @@ -175,23 +186,19 @@ async def invoke( kwargs = dict(self._client_kwargs) kwargs.update(input_kwargs) + contains_image = self._check_for_image(messages) + if contains_image: + # Currently instructor does not support streaming with multimodal + # https://github.com/567-labs/instructor/issues/1872 + kwargs["stream"] = False + if response_model is not None: if allow_partial and isinstance(response_model, BaseModel): response_model = Partial[response_model] kwargs["response_model"] = response_model # check if any of the messages contain images - elif response_model is None and ImageResponse: - for message in messages: - if not isinstance(message["content"], list): - continue - for item in message["content"]: - if not isinstance(item, Image): - continue - kwargs["response_model"] = ImageResponse - # Currently instructor does not support streaming with multimodal - # https://github.com/567-labs/instructor/issues/1872 - kwargs["stream"] = False - break + elif response_model is None and contains_image: + kwargs["response_model"] = ImageResponse output = await self.run_client(model_spec, messages, **kwargs) if output is None or output == "": From 69489fb2ee694a68e3dd201ecd7874a7d8da9eec Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 16:22:00 -0700 Subject: [PATCH 05/10] support panes --- lumen/ai/llm.py | 37 +++++++++++++++++++++++++++++-------- 1 file changed, 29 insertions(+), 8 deletions(-) diff --git a/lumen/ai/llm.py b/lumen/ai/llm.py index 268290bbe..fd1d09dfb 100644 --- a/lumen/ai/llm.py +++ b/lumen/ai/llm.py @@ -1,9 +1,11 @@ from __future__ import annotations import asyncio +import base64 import os from functools import partial +from pathlib import Path from types import SimpleNamespace from typing import Any, Literal, TypedDict @@ -130,16 +132,35 @@ def _add_system_message( messages = [{"role": "system", "content": system}] + messages return messages, input_kwargs - def _check_for_image(self, messages: list[Message]) -> bool: - for message in messages: + def _serialize_image_pane(self, image: pn.pane.image.ImageBase | Image) -> Image: + if isinstance(image, Image): + return image + + image_object = image.object + if isinstance(image_object, bytes): + # convert bytes to base64 string + base64_str = base64.b64encode(image_object).decode('utf-8') + image = Image.from_raw_base64(base64_str) + elif isinstance(image_object, (Path, str)) and Path(image_object).is_file(): + image = Image.from_path(image_object) + elif isinstance(image_object, str): + image = Image.from_url(image_object) + return image + + def _check_for_image(self, messages: list[Message]) -> tuple[list[Message], bool]: + contains_image = False + for i, message in enumerate(messages): content = message.get("content") - if isinstance(content, Image): - return True + if isinstance(content, (Image, pn.pane.image.ImageBase)): + messages[i]["content"] = self._serialize_image_pane(content) + contains_image = True + elif isinstance(content, list): for item in content: - if isinstance(item, Image): - return True - return False + if isinstance(item, (Image, pn.pane.image.ImageBase)): + messages[i]["content"] = self._serialize_image_pane(item) + contains_image = True + return messages, contains_image @classmethod def warmup(cls, model_kwargs: dict | None): @@ -186,7 +207,7 @@ async def invoke( kwargs = dict(self._client_kwargs) kwargs.update(input_kwargs) - contains_image = self._check_for_image(messages) + messages, contains_image = self._check_for_image(messages) if contains_image: # Currently instructor does not support streaming with multimodal # https://github.com/567-labs/instructor/issues/1872 From 5ddb832668a9f89e06f49ddc11b45427aca2b1e5 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 15:10:05 -0700 Subject: [PATCH 06/10] Export image --- lumen/ai/report.py | 46 +++++++++++++++++----------------------------- 1 file changed, 17 insertions(+), 29 deletions(-) diff --git a/lumen/ai/report.py b/lumen/ai/report.py index 0fc3873f7..479312d9d 100644 --- a/lumen/ai/report.py +++ b/lumen/ai/report.py @@ -2,6 +2,7 @@ import asyncio import io +import os import tempfile import traceback as tb @@ -45,7 +46,7 @@ describe_data, extract_block_source, get_block_names, wrap_logfire_on_method, ) -from .views import LumenOutput +from .views import LumenOutput, VegaLiteOutput class Task(Viewer): @@ -834,7 +835,6 @@ def _generate_sections(self, doc) -> list[dict]: sections = [] - breakpoint() for section in self._tasks: if not isinstance(section, Section): continue @@ -848,7 +848,7 @@ def _generate_sections(self, doc) -> list[dict]: # Process section outputs to find visualizations and captions image_found = False for i, out in enumerate(section.outputs): - if isinstance(out, LumenOutput) and not image_found: + if isinstance(out, VegaLiteOutput) and not image_found: # Convert LumenOutput to image image_path = self._output_to_image(out) if image_path: @@ -881,37 +881,25 @@ def _output_to_image(self, output: LumenOutput) -> str | None: str | None Path to temporary image file, or None if conversion failed """ + # Create a temporary file for the image + tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False) + tmp_path = tmp.name + tmp.close() try: - # Create a temporary file for the image - tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False) - tmp_path = tmp.name - tmp.close() - # Render the component and save as image component = output.component - - if isinstance(component, (View, Pipeline)): - # Render the view/pipeline to a panel - panel = output.render() - if hasattr(panel, 'save'): - panel.save(tmp_path) - return tmp_path - elif hasattr(component, 'save'): - component.save(tmp_path) + with open(tmp_path, 'wb') as f: + vega_pane = component.__panel__()._pane + vega_pane.param.update( + width=650, + height=400, + ) + image_bytes = vega_pane.export("png", scale=2, ppi=144) + f.write(image_bytes) return tmp_path - else: - print("Output component is not renderable to image.") - - # TODO: add vega-lite export and table to docx table - breakpoint() - - # If we couldn't save, remove the temp file - import os - os.unlink(tmp_path) - return None - except Exception as e: - print(f"Failed to convert output to image: {e}") + self.param.warning(f"Failed to convert output to image: {e}") + os.unlink(tmp_path) return None From dda33d6c39a6500483e98db418c99cbc95ed6923 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Tue, 28 Oct 2025 16:43:56 -0700 Subject: [PATCH 07/10] cleanup --- lumen/ai/report.py | 19 +++++++++---------- 1 file changed, 9 insertions(+), 10 deletions(-) diff --git a/lumen/ai/report.py b/lumen/ai/report.py index 479312d9d..987978814 100644 --- a/lumen/ai/report.py +++ b/lumen/ai/report.py @@ -40,7 +40,6 @@ ) from .llm import Llm from .memory import _Memory -from .schemas import get_metaset from .tools import FunctionTool, Tool from .utils import ( describe_data, extract_block_source, get_block_names, @@ -614,17 +613,17 @@ class Report(TaskGroup): docx_context = param.Dict(default={}, doc=""" Context dictionary for docx template rendering. If keys are not provided, the following defaults will be used: - + - 'title': self.title or 'Lumen Report' - 'subtitle': 'Generated on {date}' (e.g., 'Generated on October 27, 2025') - 'cover_page_header': '' (empty string) - 'cover_page_footer': '' (empty string) - 'content_page_header': '' (empty string) - + The following keys are always auto-generated and cannot be overridden: - 'sections': List of section dicts with 'title', 'image', and 'caption' - 'page_break': R('\f') for page breaks - + Example: report.docx_context = { 'title': 'Q4 Sales Report', @@ -753,12 +752,12 @@ def __panel__(self): def to_docx(self) -> io.BytesIO: """ Export the report to a Word document (.docx) format. - + Returns ------- BytesIO A BytesIO buffer containing the rendered docx document. - + Raises ------ RuntimeError @@ -819,12 +818,12 @@ def to_docx(self) -> io.BytesIO: def _generate_sections(self, doc) -> list[dict]: """ Generate sections list from report tasks for docx template. - + Arguments --------- doc : DocxTemplate The document template instance (needed for InlineImage creation) - + Returns ------- list[dict] @@ -870,12 +869,12 @@ def _generate_sections(self, doc) -> list[dict]: def _output_to_image(self, output: LumenOutput) -> str | None: """ Convert a LumenOutput to an image file path. - + Arguments --------- output : LumenOutput The output to convert - + Returns ------- str | None From d86cc279339256445d5a4498129eb97c6edb365b Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Fri, 31 Oct 2025 10:37:22 -0700 Subject: [PATCH 08/10] address issues --- lumen/ai/export.py | 203 ++++++++++++++++++++++++++++++++++++++++++++- lumen/ai/report.py | 167 +++++-------------------------------- pyproject.toml | 2 +- 3 files changed, 223 insertions(+), 149 deletions(-) diff --git a/lumen/ai/export.py b/lumen/ai/export.py index 63daf15c2..137d7a291 100644 --- a/lumen/ai/export.py +++ b/lumen/ai/export.py @@ -1,22 +1,32 @@ import base64 +import io import os import re +import tempfile +import warnings +from datetime import datetime from io import BytesIO +from pathlib import Path from textwrap import dedent from typing import Any import nbformat import yaml +from docx.shared import Mm +from docxtpl import ( + DocxTemplate, InlineImage, R, RichText, +) from panel import Column from panel.chat import ChatMessage, ChatStep from panel.pane.image import ImageBase +from panel_material_ui import Typography from ..config import config from ..pipeline import Pipeline from ..views import View -from .views import LumenOutput +from .views import LumenOutput, VegaLiteOutput def make_md_cell(text: str): @@ -135,3 +145,194 @@ def export_notebook(messages: list[ChatMessage], preamble: str = ""): cells, extensions = render_cells(messages) cells = make_preamble(preamble, extensions=extensions) + cells return write_notebook(cells) + + +def render_docx_template( + outputs: list, + docx_template_path: str | Path, + **docx_context: dict, +) -> io.BytesIO: + """ + Export outputs to a Word document (.docx) format. + + Arguments + --------- + outputs : list + List of output objects to export (Typography, LumenOutput, etc.) + docx_context : dict | None + Context dictionary for docx template rendering. If keys are not provided, + the following defaults will be used: + + - 'title': title parameter or 'Lumen Report' + - 'subtitle': 'Generated on {date}' (e.g., 'Generated on October 27, 2025') + - 'cover_page_header': '' (empty string) + - 'cover_page_footer': '' (empty string) + - 'content_page_header': '' (empty string) + + The following keys are always auto-generated and cannot be overridden: + - 'sections': List of section dicts with 'title', 'image', and 'caption' + - 'page_break': R('\f') for page breaks + docx_template_path : str | None + Path to the docx template file. If None, uses the default Lumen template. + + Returns + ------- + BytesIO + A BytesIO buffer containing the rendered docx document. + + Raises + ------ + RuntimeError + If the outputs list is empty. + FileNotFoundError + If the template file is not found. + + Example + ------- + buffer = to_docx( + outputs=report.outputs, + **{ + 'subtitle': 'Quarterly Analysis', + 'cover_page_header': 'ACME Corporation' + } + ) + """ + # Set default template path + if docx_template_path is None: + docx_template_path = str(Path(__file__).parent / "assets" / "lumen_template.docx") + + # Load template + template_path = Path(docx_template_path) + if not template_path.exists(): + raise FileNotFoundError(f"Template file not found: {template_path}") + + doc = DocxTemplate(str(template_path)) + + # Start with copy of docx_context or empty dict + context = dict(docx_context) if docx_context else {} + + # Set defaults for missing keys + if 'title' not in context: + context['title'] = "Lumen Report" + + if 'subtitle' not in context: + date_string = datetime.now().strftime("%B %d, %Y") + context['subtitle'] = f"Generated on {date_string}" + + if 'cover_page_header' not in context: + context['cover_page_header'] = "" + + if 'cover_page_footer' not in context: + context['cover_page_footer'] = "" + + if 'content_page_header' not in context: + context['content_page_header'] = "" + + # Always generate sections from outputs + context['sections'] = generate_sections(doc, outputs) + + # Always set page_break + context['page_break'] = R("\f") + + # Render template + doc.render(context) + + # Return as BytesIO + buffer = io.BytesIO() + doc.save(buffer) + buffer.seek(0) + return buffer + + +def generate_sections(doc, outputs: list) -> list[dict]: + """ + Generate sections list from outputs for docx template. + + Arguments + --------- + doc : DocxTemplate + The document template instance (needed for InlineImage creation) + outputs : list + List of output objects to process + + Returns + ------- + list[dict] + List of section dictionaries with title, image, and caption + """ + sections = [] + current_section = None + skip_next = False + + for i, out in enumerate(outputs): + if skip_next: + skip_next = False + continue + + # Check if this is a section header (h2 Typography) + if isinstance(out, Typography) and out.variant == 'h2': + # Save previous section if it has an image + if current_section and current_section.get('image'): + sections.append(current_section) + + # Start new section + current_section = { + "title": out.object or "Untitled Section", + "image": None, + "caption": RichText("") + } + + # Check for VegaLiteOutput to use as image + elif isinstance(out, VegaLiteOutput) and current_section and not current_section.get('image'): + image_path = output_to_image(out) + if image_path: + current_section["image"] = InlineImage(doc, image_path, width=Mm(160)) + + # Check if next output is a Typography for caption + if i + 1 < len(outputs): + next_out = outputs[i + 1] + if isinstance(next_out, Typography): + current_section["caption"] = RichText(next_out.object) + skip_next = True # Skip the caption in next iteration + + # Add the last section if it has an image + if current_section and current_section.get('image'): + sections.append(current_section) + + return sections + + +def output_to_image(output: VegaLiteOutput) -> str | None: + """ + Convert a VegaLiteOutput to an image file path. + + Arguments + --------- + output : VegaLiteOutput + The output to convert + + Returns + ------- + str | None + Path to temporary image file, or None if conversion failed + """ + # Create a temporary file for the image + tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False) + tmp_path = tmp.name + tmp.close() + try: + # Render the component and save as image + component = output.component + with open(tmp_path, 'wb') as f: + vega_pane = component.__panel__()._pane + vega_pane.param.update( + width=650, + height=400, + ) + image_bytes = vega_pane.export("png", scale=2, ppi=144) + f.write(image_bytes) + return tmp_path + except Exception as e: + warnings.warn(f"Failed to convert output to image: {e}", stacklevel=2) + os.unlink(tmp_path) + return None diff --git a/lumen/ai/report.py b/lumen/ai/report.py index 987978814..18141bfd5 100644 --- a/lumen/ai/report.py +++ b/lumen/ai/report.py @@ -2,13 +2,10 @@ import asyncio import io -import os -import tempfile import traceback as tb from abc import abstractmethod from collections.abc import Iterator -from datetime import datetime from functools import partial from pathlib import Path from types import FunctionType @@ -36,7 +33,8 @@ from .agents import AnalystAgent from .config import MissingContextError from .export import ( - format_output, make_md_cell, make_preamble, write_notebook, + format_output, make_md_cell, make_preamble, render_docx_template, + write_notebook, ) from .llm import Llm from .memory import _Memory @@ -45,7 +43,7 @@ describe_data, extract_block_source, get_block_names, wrap_logfire_on_method, ) -from .views import LumenOutput, VegaLiteOutput +from .views import LumenOutput class Task(Viewer): @@ -740,15 +738,6 @@ async def _run_task(self, i: int, task: Section, **kwargs): task.param.unwatch(watcher) return outputs - def __panel__(self): - return Column( - self._menu, - Container( - self._container, sizing_mode="stretch_both", height_policy="max", - stylesheets=[":host > div { overflow-y: auto; }"], min_height=600 - ) - ) - def to_docx(self) -> io.BytesIO: """ Export the report to a Word document (.docx) format. @@ -765,141 +754,25 @@ def to_docx(self) -> io.BytesIO: FileNotFoundError If the template file is not found. """ - from docxtpl import DocxTemplate, R - - # Validate execution - if len(self) and not len(self.outputs): + if not len(self.outputs): raise RuntimeError( "Report has not been executed, run report before exporting to_docx." ) - # Load template - template_path = Path(self.docx_template_path) - if not template_path.exists(): - raise FileNotFoundError(f"Template file not found: {template_path}") - - doc = DocxTemplate(str(template_path)) - - # Start with copy of docx_context - context = dict(self.docx_context) - - # Set defaults for missing keys - if 'title' not in context: - context['title'] = self.title or "Lumen Report" - - if 'subtitle' not in context: - date_string = datetime.now().strftime("%B %d, %Y") - context['subtitle'] = f"Generated on {date_string}" - - if 'cover_page_header' not in context: - context['cover_page_header'] = "" - - if 'cover_page_footer' not in context: - context['cover_page_footer'] = "" - - if 'content_page_header' not in context: - context['content_page_header'] = "" - - # Always generate sections from report outputs - context['sections'] = self._generate_sections(doc) - - # Always set page_break - context['page_break'] = R("\f") - - # Render template - doc.render(context) - - # Return as BytesIO - buffer = io.BytesIO() - doc.save(buffer) - buffer.seek(0) - return buffer - - def _generate_sections(self, doc) -> list[dict]: - """ - Generate sections list from report tasks for docx template. - - Arguments - --------- - doc : DocxTemplate - The document template instance (needed for InlineImage creation) - - Returns - ------- - list[dict] - List of section dictionaries with title, image, and caption - """ - from docx.shared import Mm - from docxtpl import InlineImage, RichText - - sections = [] - - for section in self._tasks: - if not isinstance(section, Section): - continue - - section_dict = { - "title": section.title or "Untitled Section", - "image": None, - "caption": RichText("") - } - - # Process section outputs to find visualizations and captions - image_found = False - for i, out in enumerate(section.outputs): - if isinstance(out, VegaLiteOutput) and not image_found: - # Convert LumenOutput to image - image_path = self._output_to_image(out) - if image_path: - section_dict["image"] = InlineImage(doc, image_path, width=Mm(160)) - image_found = True - - # Check if next output is a Typography for caption - if i + 1 < len(section.outputs): - next_out = section.outputs[i + 1] - if isinstance(next_out, Typography): - section_dict["caption"] = RichText(next_out.object) - break - - if section_dict["image"]: # Only add section if it has an image - sections.append(section_dict) - - return sections - - def _output_to_image(self, output: LumenOutput) -> str | None: - """ - Convert a LumenOutput to an image file path. - - Arguments - --------- - output : LumenOutput - The output to convert + return render_docx_template( + self.outputs, + self.docx_template_path, + **self.docx_context, + ) - Returns - ------- - str | None - Path to temporary image file, or None if conversion failed - """ - # Create a temporary file for the image - tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False) - tmp_path = tmp.name - tmp.close() - try: - # Render the component and save as image - component = output.component - with open(tmp_path, 'wb') as f: - vega_pane = component.__panel__()._pane - vega_pane.param.update( - width=650, - height=400, - ) - image_bytes = vega_pane.export("png", scale=2, ppi=144) - f.write(image_bytes) - return tmp_path - except Exception as e: - self.param.warning(f"Failed to convert output to image: {e}") - os.unlink(tmp_path) - return None + def __panel__(self): + return Column( + self._menu, + Container( + self._container, sizing_mode="stretch_both", height_policy="max", + stylesheets=[":host > div { overflow-y: auto; }"], min_height=600 + ) + ) class Action(Task): @@ -945,7 +818,7 @@ class SQLQuery(Action): template_overrides = param.Dict(default={}, doc=""" Template overrides to provide to the AnalystAgent.""") - user_content = param.String(default=None, doc=""" + analyst_instructions = param.String(default=None, doc=""" Instructions to provide to the analyst agent, i.e. what to focus on; if unset no additional instructions are provided.""") @@ -1008,9 +881,9 @@ async def _execute(self, **kwargs): self.memory["table"] = self.table out = LumenOutput(component=pipeline) outputs = [Typography(f"### {self.title}", variant='h4', margin=(10, 10, 0, 10)), out] if self.title else [out] - if self.user_content: + if self.analyst_instructions: caption = await AnalystAgent(llm=self.llm, template_overrides=self.template_overrides).respond( - [{"role": "user", "content": self.user_content}] + [{"role": "user", "content": self.analyst_instructions}] ) outputs.append(Typography(caption.object)) return outputs diff --git a/pyproject.toml b/pyproject.toml index ce9621e8c..19094ab68 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -50,7 +50,7 @@ tests = ['pytest', 'pytest-rerunfailures', 'pytest-asyncio'] sql = ['duckdb', 'intake-sql', 'sqlalchemy'] ai = [ 'griffe', 'nbformat', 'duckdb >= 1.2.0', 'pyarrow', 'instructor >=1.6.4', 'pydantic >=2.8.0', 'pydantic-extra-types', 'panel-graphic-walker[kernel] >=0.6.4', - 'markitdown', 'semchunk', 'tiktoken', 'chardet', "panel-material-ui >=0.4.0", "tabulate" + 'markitdown', 'semchunk', 'tiktoken', 'chardet', "panel-material-ui >=0.4.0", "tabulate", "docxtpl" ] ai-local = ['lumen[ai]', 'huggingface_hub', 'hf_xet'] ai-openai = ['lumen[ai]', 'openai'] From 2dc7f1f7425f03d465769484a74caeaf111d3609 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Fri, 31 Oct 2025 10:38:34 -0700 Subject: [PATCH 09/10] clarify method --- lumen/ai/export.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/lumen/ai/export.py b/lumen/ai/export.py index 137d7a291..8366ac2dc 100644 --- a/lumen/ai/export.py +++ b/lumen/ai/export.py @@ -229,7 +229,7 @@ def render_docx_template( context['content_page_header'] = "" # Always generate sections from outputs - context['sections'] = generate_sections(doc, outputs) + context['sections'] = generate_docx_sections(doc, outputs) # Always set page_break context['page_break'] = R("\f") @@ -244,7 +244,7 @@ def render_docx_template( return buffer -def generate_sections(doc, outputs: list) -> list[dict]: +def generate_docx_sections(doc, outputs: list) -> list[dict]: """ Generate sections list from outputs for docx template. From ad076a9567a2a1c363069da0fc15886dc58f8884 Mon Sep 17 00:00:00 2001 From: Andrew Huang Date: Fri, 31 Oct 2025 11:25:54 -0700 Subject: [PATCH 10/10] fix migration --- lumen/ai/export.py | 86 ++++++++++++++++++++-------------------------- lumen/ai/report.py | 2 +- 2 files changed, 38 insertions(+), 50 deletions(-) diff --git a/lumen/ai/export.py b/lumen/ai/export.py index 8366ac2dc..443903ab7 100644 --- a/lumen/ai/export.py +++ b/lumen/ai/export.py @@ -148,7 +148,7 @@ def export_notebook(messages: list[ChatMessage], preamble: str = ""): def render_docx_template( - outputs: list, + sections: list, docx_template_path: str | Path, **docx_context: dict, ) -> io.BytesIO: @@ -157,8 +157,8 @@ def render_docx_template( Arguments --------- - outputs : list - List of output objects to export (Typography, LumenOutput, etc.) + sections: list + List of sections to process docx_context : dict | None Context dictionary for docx template rendering. If keys are not provided, the following defaults will be used: @@ -228,8 +228,8 @@ def render_docx_template( if 'content_page_header' not in context: context['content_page_header'] = "" - # Always generate sections from outputs - context['sections'] = generate_docx_sections(doc, outputs) + # Always generate sections + context['sections'] = generate_docx_sections(doc, sections) # Always set page_break context['page_break'] = R("\f") @@ -244,62 +244,50 @@ def render_docx_template( return buffer -def generate_docx_sections(doc, outputs: list) -> list[dict]: +def generate_docx_sections(doc: DocxTemplate, sections: list) -> list[dict]: """ - Generate sections list from outputs for docx template. + Generate sections list from report tasks for docx template. Arguments --------- doc : DocxTemplate The document template instance (needed for InlineImage creation) - outputs : list - List of output objects to process + sections : list + List of sections to process Returns ------- list[dict] List of section dictionaries with title, image, and caption """ - sections = [] - current_section = None - skip_next = False - - for i, out in enumerate(outputs): - if skip_next: - skip_next = False - continue - - # Check if this is a section header (h2 Typography) - if isinstance(out, Typography) and out.variant == 'h2': - # Save previous section if it has an image - if current_section and current_section.get('image'): - sections.append(current_section) - - # Start new section - current_section = { - "title": out.object or "Untitled Section", - "image": None, - "caption": RichText("") - } - - # Check for VegaLiteOutput to use as image - elif isinstance(out, VegaLiteOutput) and current_section and not current_section.get('image'): - image_path = output_to_image(out) - if image_path: - current_section["image"] = InlineImage(doc, image_path, width=Mm(160)) - - # Check if next output is a Typography for caption - if i + 1 < len(outputs): - next_out = outputs[i + 1] - if isinstance(next_out, Typography): - current_section["caption"] = RichText(next_out.object) - skip_next = True # Skip the caption in next iteration - - # Add the last section if it has an image - if current_section and current_section.get('image'): - sections.append(current_section) - - return sections + output_sections = [] + for section in sections: + section_dict = { + "title": section.title or "Untitled Section", + "image": None, + "caption": RichText("") + } + + # Process section outputs to find visualizations and captions + image_found = False + for i, out in enumerate(section.outputs): + if isinstance(out, VegaLiteOutput) and not image_found: + # Convert LumenOutput to image + image_path = output_to_image(out) + if image_path: + section_dict["image"] = InlineImage(doc, image_path, width=Mm(160)) + image_found = True + + # Check if next output is a Typography for caption + if i + 1 < len(section.outputs): + next_out = section.outputs[i + 1] + if isinstance(next_out, Typography): + section_dict["caption"] = RichText(next_out.object) + break + + if section_dict["image"]: # Only add section if it has an image + output_sections.append(section_dict) + return output_sections def output_to_image(output: VegaLiteOutput) -> str | None: diff --git a/lumen/ai/report.py b/lumen/ai/report.py index 18141bfd5..fa619a04b 100644 --- a/lumen/ai/report.py +++ b/lumen/ai/report.py @@ -760,7 +760,7 @@ def to_docx(self) -> io.BytesIO: ) return render_docx_template( - self.outputs, + [task for task in self._tasks if isinstance(task, Section)], self.docx_template_path, **self.docx_context, )